Skip to main content

rustynes_ppu/
ppu.rs

1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: this PPU contains code derived from Mesen2 (GPL-3.0-or-later): the sprite-evaluation FSM and OAM-data-bus model, `Core/NES/NesPpu.cpp` (`ProcessSpriteEvaluation` / `ReadSpriteRam`), and the optional OAM-decay model (`ReadSpriteRam` / `WriteSpriteRam`, `OamDecayCycleCount`; added v2.1.4, disclosed v2.9.4); it also incorporates models ported from TriCNES (MIT) — the ALE / octal-latch address-multiplex and the OAM-corruption behavior. See docs/originality-and-provenance.md (Section 1)
4// and NOTICE for the complete, audited derivation record.
5//! 2C02 PPU core: state, register surface, scanline counter, NMI signaling.
6//!
7//! See `docs/ppu-2c02.md`. Background and sprite *rendering* (per-dot tile
8//! fetch, shift registers, sprite evaluation, sprite-zero hit) is plumbed
9//! through this struct but the visible-pixel output path is filled in by
10//! Sprints 2-2 and 2-3 — the surface and scanline FSM here is what
11//! Sprint 2-1 delivers.
12
13use crate::bus::{BgSplitState, ExAttribute, PpuBus};
14use crate::palette::{build_rgba_lut, build_rgba_lut_from_base};
15use crate::registers::{PpuCtrl, PpuMask, PpuStatus};
16use alloc::boxed::Box;
17use alloc::vec;
18
19/// Visible screen width in pixels, and the stride of every per-pixel buffer the
20/// PPU exposes.
21///
22/// v2.3.8 — the single definition. `provenance::SCREEN_W` was a second copy of
23/// this number in a `debug-hooks`-gated module, which made the width
24/// unreachable from ungated code and invited a third copy rather than a
25/// dependency. It now aliases this.
26pub const SCREEN_WIDTH: usize = 256;
27
28/// Visible screen height in pixels. Companion to [`SCREEN_WIDTH`].
29pub const SCREEN_HEIGHT: usize = 240;
30
31/// RGBA8 framebuffer length in bytes (256 × 240 × 4).
32pub const FRAMEBUFFER_LEN: usize = SCREEN_WIDTH * SCREEN_HEIGHT * 4;
33
34/// Visible pixel count (256 × 240) — length of the parallel
35/// [`Ppu::index_framebuffer`] (one `u16` per pixel).
36pub const FRAMEBUFFER_PIXELS: usize = SCREEN_WIDTH * SCREEN_HEIGHT;
37
38/// v3.1.0 (`T-SPRITE-LIMIT`) — the most sprites the "disable sprite limit"
39/// option can add to one scanline: all 64 OAM entries minus the eight the
40/// hardware draws.
41pub const MAX_EXTRA_SPRITES: usize = 64 - 8;
42
43/// v1.2.0 beta.2 (Workstream C3) — per-pixel HD-pack tile-source record.
44///
45/// One entry per visible pixel (parallel to [`Ppu::index_framebuffer`]),
46/// populated in the pixel-emit path only when the `hd-pack` cargo feature is
47/// enabled. It records the **identity of the CHR tile** that produced the
48/// pixel — the 16-byte pattern-table tile base address (in PPU `$0000..=$1FFF`
49/// pattern space, fine-Y masked off), the 2-bit attribute/sprite palette, the
50/// sprite flip flags, and whether the source was a sprite or the background.
51///
52/// It is pure output telemetry: it mirrors data the renderer already computed,
53/// reads no new VRAM, and changes no emulation state — so it is byte-identical
54/// with the feature on or off and is never serialized into a save-state. The
55/// frontend's Mesen-style HD-pack loader groups these by 8×8 screen cell, hashes
56/// the referenced CHR bytes, and substitutes hi-res replacement tiles at blit
57/// time. See `docs/ppu-2c02.md` §HD-pack tile-source export.
58#[cfg(feature = "hd-pack")]
59#[derive(Clone, Copy, Debug, Eq, PartialEq)]
60pub struct HdTileSource {
61    /// 16-byte CHR tile base address in pattern space (`$0000..=$1FF0`, low
62    /// nibble always 0). `tile = (addr >> 4) & 0xFF`, `table = addr & 0x1000`.
63    /// `0xFFFF` marks a transparent / universal-background pixel (no tile).
64    pub chr_addr: u16,
65    /// Final palette group: BG attribute (0..=3) or sprite palette (0..=3).
66    pub palette: u8,
67    /// `true` when the pixel came from a sprite (vs. the background).
68    pub is_sprite: bool,
69    /// Sprite horizontal flip (always `false` for BG pixels).
70    pub flip_h: bool,
71    /// Sprite vertical flip (always `false` for BG pixels).
72    pub flip_v: bool,
73    /// Mesen `PaletteColors`: the tile's active palette packed as the HD-pack
74    /// tile-identity key expects. BG = `pr[base+3] | pr[base+2]<<8 |
75    /// pr[base+1]<<16 | pr[0]<<24`; sprite = `0xFF000000 | pr[base+3] |
76    /// pr[base+2]<<8 | pr[base+1]<<16` (top byte `0xFF` = the sprite/BG
77    /// discriminator). Part of the CHR-RAM/CHR-ROM tile key (HdNesPpu.h:119/167).
78    pub palette_colors: u32,
79    /// Which texel COLUMN (0..=7) of the 8x8 tile this screen pixel samples —
80    /// Mesen `OffsetX` (HdNesPpu.h:172). For BG it folds in fine-X scroll
81    /// (`(fineX + (pixel_x & 7)) & 7`); for a sprite it is the column within the
82    /// sprite tile. The HD compositor samples the replacement at this column so
83    /// the high-def tile tracks the scrolled / sprite position pixel-for-pixel
84    /// (flips are applied at sample time). Output-only.
85    pub offset_x: u8,
86    /// Which texel ROW (0..=7) of the 8x8 tile this screen pixel samples — Mesen
87    /// `OffsetY`. BG = fine-Y; sprite = the row within the sprite tile.
88    pub offset_y: u8,
89    /// Mesen CHR-ROM `TileIndex`: the ABSOLUTE post-banking CHR-ROM tile number
90    /// (`chr_phys(addr) / 16`) for a CHR-ROM cart, or [`HD_CHR_RAM`] when CHR is
91    /// RAM (the tile is content-hashed instead). The HD-pack key uses
92    /// `TileIndex ^ PaletteColors` for CHR-ROM and `CalculateHash(palette ++
93    /// data)` for CHR-RAM (Mesen `HdTileKey`). Captured at fetch (the only point
94    /// with mapper access). Output-only.
95    pub chr_tile_index: u32,
96    /// v1.8.9 — the `$2001` grayscale + emphasis bits at this pixel
97    /// (`mask.bits() & 0xE1`: bit 0 = grayscale, bits 5-7 = R/G/B emphasis). The
98    /// HD compositor re-applies them to the replacement texel (Mesen
99    /// `ProcessGrayscaleAndEmphasis`) so HD tiles track grayscale / emphasis fades
100    /// like the base frame, which already has them baked in. Output-only.
101    pub color_mask: u8,
102    /// v1.8.9 — every opaque sprite covering this pixel (up to 4), front-to-back,
103    /// for Mesen `spriteAtPosition` / `spriteNearby` (which match ANY covering
104    /// sprite, including ones a higher-priority BG occludes). `sprites[0]` is the
105    /// front-most. The existing `is_sprite` + tile fields still describe the
106    /// VISIBLE pixel (the winning layer); these add the hidden layers. Output-only.
107    pub sprites: [HdSprite; 4],
108    /// Number of valid entries in [`Self::sprites`] (`0..=4`).
109    pub sprite_count: u8,
110}
111
112/// One sprite covering a pixel, for the HD-pack multi-sprite conditions
113/// (`spriteAtPosition` / `spriteNearby`). Carries just the identity those
114/// conditions match on. Output-only telemetry.
115#[cfg(feature = "hd-pack")]
116#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
117pub struct HdSprite {
118    /// Absolute CHR-ROM tile index (`chr_phys/16`), or [`HD_CHR_RAM`] for CHR-RAM.
119    pub chr_tile_index: u32,
120    /// The sprite's packed `PaletteColors` (Mesen key form).
121    pub palette_colors: u32,
122}
123
124/// `chr_tile_index` sentinel meaning "CHR is RAM" — the tile is keyed by its 16
125/// CHR bytes (content) rather than by an absolute CHR-ROM tile index.
126#[cfg(feature = "hd-pack")]
127pub const HD_CHR_RAM: u32 = u32::MAX;
128
129#[cfg(feature = "hd-pack")]
130impl Default for HdTileSource {
131    /// A blank record: no tile, and the CHR-RAM sentinel (so an unwritten /
132    /// default record is content-keyed, never mistaken for CHR-ROM tile 0).
133    fn default() -> Self {
134        Self {
135            chr_addr: 0,
136            palette: 0,
137            is_sprite: false,
138            flip_h: false,
139            flip_v: false,
140            palette_colors: 0,
141            offset_x: 0,
142            offset_y: 0,
143            chr_tile_index: HD_CHR_RAM,
144            color_mask: 0,
145            sprites: [HdSprite::default(); 4],
146            sprite_count: 0,
147        }
148    }
149}
150
151/// Sentinel `chr_addr` for a transparent / universal-background HD-pack pixel.
152#[cfg(feature = "hd-pack")]
153pub const HD_TILE_NONE: u16 = 0xFFFF;
154
155/// Diagnostic: capture the (frame, scanline, dot, mask) of `$2007` reads to
156/// pin where the `$2007 Stress` test's per-dot reads land vs the visible
157/// scanline they target. Gated; default build unaffected.
158pub mod read2007_diag {
159    use core::sync::atomic::AtomicU32;
160    /// Next free slot (count of captured `$2007` reads).
161    pub static IDX: AtomicU32 = AtomicU32::new(0);
162    /// Packed: `((scanline+1)<<18) | (dot<<5) | (rendering_enabled<<1) | is_render`.
163    pub static LOG: [AtomicU32; 1024] = [const { AtomicU32::new(0) }; 1024];
164    /// Tunable PPU-dot countdown: `$2007` read-during-rendering to reload.
165    ///
166    /// The `PPUDATA` state machine reloads `data_buffer` from the fetch cadence
167    /// (`TriCNES` `PPU_DATA_StateMachine` latch cascade: read-end -> ALE +2 ->
168    /// reload +4 dots; with R1's fixed CPU<->PPU mod-4 phase the landing is a
169    /// constant offset from the register-read sample point). 0 = immediate.
170    /// Default 5 = the empirical winner (W2 sweep 0-12: 5 -> 170/170 stable
171    /// reads on the `$2007 Stress` answer key; 6/7 -> 169; everything else
172    /// far below). Env knob `RUSTYNES_2007_DELAY` (wired in the diag bins).
173    pub static RENDER_BUFFER_DOT_DELAY: AtomicU32 = AtomicU32::new(5);
174    /// Sub-knob: defer the `$2007` v-glitch increment to the `TStep` dot.
175    ///
176    /// The `TStep` is the SAME dot as the buffer reload — `TriCNES`
177    /// `PPU_DATA_StateMachine_Half`: `PPU_2007_TStep = TStep_Latch || PD_RB`
178    /// — instead of read time. With the immediate increment, every fetch in
179    /// the read-to-reload window uses the post-glitch `v` (coarse-x +1 -> NT
180    /// byte = tile+1; fine-y +1 -> PT row+1) — exactly the per-index mismatch
181    /// signature the W2 baseline measured. 1 = deferred (default), 0 =
182    /// immediate (legacy). Env knob `RUSTYNES_2007_VINC` (wired in the diag
183    /// bins).
184    pub static RENDER_BUFFER_DEFER_V_INC: AtomicU32 = AtomicU32::new(1);
185}
186
187/// v2.0.3 (ADR 0030) octal-latch calibration tracer.
188///
189/// Entirely a diagnostic: a lock-free ring of packed per-event records captured
190/// only when `ENABLE` is set (via the env knob wired in the test harness). It
191/// costs nothing in an untraced run — each `push` call site is a single relaxed
192/// atomic load + early-return branch, and every call site sits in a cold
193/// corruption branch (never the steady-state per-dot path). Records the
194/// corruption-relevant events (`$2006`/`$2007` during render, hybrid/stale
195/// splices) on scanlines 2-5 so a run can be cross-diffed against the `TriCNES`
196/// per-dot bus sequence.
197///
198/// Gated behind the default-off `ppu-octal-trace` dev feature. Behavior never
199/// depends on it — the 2-cycle-ALE fetch model (now the only PPU fetch path, ADR
200/// 0030) drives its `push` call sites unconditionally, but with the feature off
201/// `push` is a zero-cost no-op ([`stub`](self)) and the ring's 64-bit-atomic
202/// storage does not exist, so the shipped hot path is untouched and the
203/// `#![no_std]` chip stack (whose `thumbv7em` target lacks 64-bit atomics) still
204/// builds. Enable with `--features ppu-octal-trace` to capture a trace.
205#[cfg(feature = "ppu-octal-trace")]
206pub mod octal_trace {
207    use core::sync::atomic::{AtomicU32, AtomicU64};
208    /// 1 = capture enabled. Off by default; an untraced flag-on run pays only a
209    /// single relaxed atomic load + branch per `push` call site.
210    pub static ENABLE: AtomicU32 = AtomicU32::new(0);
211    /// Next free slot (saturating; stops at `LOG.len()`).
212    pub static IDX: AtomicU32 = AtomicU32::new(0);
213    /// Packed record: `(kind<<58) | (frame<<44) | (scanline<<32) | (dot<<20) | value`
214    /// where `value` is event-specific (a 20-bit address / latch payload).
215    pub static LOG: [AtomicU64; 4096] = [const { AtomicU64::new(0) }; 4096];
216
217    /// Event kind: `$2006` second write during rendering (value = new `v`).
218    pub const K_W2006: u64 = 1;
219    /// Event kind: `$2007` read during rendering (value = `v`).
220    pub const K_R2007: u64 = 2;
221    /// Event kind: hybrid nametable splice fired (value = effective address).
222    pub const K_HYBRID: u64 = 3;
223    /// Event kind: stale-latch pattern splice fired (value = effective address).
224    pub const K_STALE: u64 = 4;
225    /// Event kind: `$2007` state-machine countdown landed (value = data byte).
226    pub const K_SMLAND: u64 = 5;
227
228    /// Push one record (no-op unless `ENABLE` and slots remain).
229    #[allow(clippy::cast_sign_loss)] // scanline filtered to 2..=5 (always >= 0)
230    pub fn push(kind: u64, frame: u64, scanline: i16, dot: u16, value: u32) {
231        if ENABLE.load(core::sync::atomic::Ordering::Relaxed) == 0 {
232            return;
233        }
234        // Restrict to the corruption-relevant visible scanlines (ALE+Read is
235        // scanline 3, Hybrid scanline 4) to keep the ring from overflowing on
236        // the many unrelated `$2006`/`$2007`-during-render writes elsewhere.
237        if !(2..=5).contains(&scanline) {
238            return;
239        }
240        let i = IDX.fetch_add(1, core::sync::atomic::Ordering::Relaxed) as usize;
241        if i >= LOG.len() {
242            return;
243        }
244        let sl = (u64::from((scanline & 0x0FFF) as u16)) << 32;
245        let packed = (kind << 58)
246            | ((frame & 0x3FFF) << 44)
247            | sl
248            | ((u64::from(dot) & 0xFFF) << 20)
249            | u64::from(value & 0xF_FFFF);
250        LOG[i].store(packed, core::sync::atomic::Ordering::Relaxed);
251    }
252}
253
254/// Zero-cost no-op stand-in for the octal-latch tracer.
255///
256/// Compiled when the `ppu-octal-trace` dev feature is off (the default, and the
257/// only config the `#![no_std]` chip stack builds — the full tracer's ring needs
258/// 64-bit atomics the `thumbv7em` target lacks). Exposes the same `push`
259/// signature and `K_*` event-kind constants the 2-cycle-ALE fetch path
260/// references, so those call sites stay unconditional; `push` here is an empty
261/// body the optimizer removes entirely. Behavior is thus byte-identical to the
262/// full tracer with capture disabled — the tracer only ever observes, never
263/// influences, emulation.
264#[cfg(not(feature = "ppu-octal-trace"))]
265pub mod octal_trace {
266    /// Event kind: `$2006` second write during rendering.
267    pub const K_W2006: u64 = 1;
268    /// Event kind: `$2007` read during rendering.
269    pub const K_R2007: u64 = 2;
270    /// Event kind: hybrid nametable splice fired.
271    pub const K_HYBRID: u64 = 3;
272    /// Event kind: stale-latch pattern splice fired.
273    pub const K_STALE: u64 = 4;
274    /// Event kind: `$2007` state-machine countdown landed.
275    pub const K_SMLAND: u64 = 5;
276
277    /// No-op (the `ppu-octal-trace` feature is off). Compiles to nothing.
278    #[inline(always)]
279    pub const fn push(_kind: u64, _frame: u64, _scanline: i16, _dot: u16, _value: u32) {}
280}
281
282/// v2.0.3 (ADR 0030, Option 1) — the delayed-`CopyV` countdown length in PPU
283/// dots (`TriCNES` `PPU_Update2006Delay`, `Emulator.cs:9837-9843`, which is 4
284/// for three of the four CPU/PPU sub-cycle alignments and 5 for the fourth).
285/// `RustyNES`'s lockstep bus applies the `$2006` write at the start of a CPU
286/// cycle; the corrupted nametable read is the phase-1 dot of the fetch group one
287/// coarse-X past the write. Empirically calibrated against the `TriCNES` per-dot
288/// bus trace so the countdown lands on that read after exactly one `inc_hori_v`
289/// and one phase-0 NT ALE (which loads the one-tile-ahead `$19` low byte). See
290/// the v2.0.3 campaign plan.
291const COPY_V_DELAY: u8 = 4;
292
293/// v2.1.4 F2.3 — optional OAM decay threshold, in **CPU cycles**.
294///
295/// Provenance: the OAM-decay model (this constant, the per-row timestamps, the
296/// refresh-on-read/write rule and the decayed-byte pattern in
297/// `oam_decay_on_read`) is **derived from Mesen2's `NesPpu.cpp`**
298/// (`ReadSpriteRam` / `WriteSpriteRam`, `OamDecayCycleCount`),
299/// GPL-3.0-or-later. It was written that way at v2.1.4 (#265, whose commit
300/// says so) and recorded in this file's header, NOTICE and
301/// docs/originality-and-provenance.md (Section 1) only at v2.9.4.
302///
303/// The 2C02's Object Attribute Memory is dynamic RAM: each row is implicitly
304/// refreshed every time sprite evaluation (or a `$2004` access) reads it during
305/// rendering, but with rendering disabled long enough the un-refreshed rows lose
306/// their charge and decay to a fixed garbage pattern. Mesen2 models this as a
307/// per-8-byte-row CPU-cycle timestamp with a 3000-cycle refresh window
308/// (`NesPpu::OamDecayCycleCount`, `Core/NES/NesPpu.cpp`); a read/write that lands
309/// within 3000 CPU cycles of the row's last touch refreshes it, otherwise the row
310/// has decayed. This value mirrors Mesen2's constant exactly so the two agree on
311/// when a row is considered stale.
312///
313/// This whole model is **off by default** (`Ppu::oam_decay_enabled == false`) and
314/// **NTSC/Dendy-only** — on PAL the far more frequent refresh cadence masks decay
315/// entirely, so the feature is never applied there. With it off, no OAM access
316/// consults the decay state and the PPU is byte-identical to a build that never
317/// had the field. See `docs/ppu-2c02.md` (§OAM decay).
318const OAM_DECAY_CPU_CYCLES: u64 = 3000;
319
320/// Region governs the size of the post-render-to-pre-render scanline span.
321#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)]
322pub enum PpuRegion {
323    /// NTSC (and Famicom). 262 scanlines per frame, pre-render = scanline 261.
324    Ntsc,
325    /// PAL. 312 scanlines per frame, pre-render = scanline 311.
326    Pal,
327    /// Dendy (Russian PAL famiclone). 312 scanlines, but VBL starts at 291.
328    Dendy,
329}
330
331impl PpuRegion {
332    /// Pre-render scanline number.
333    #[must_use]
334    pub const fn prerender_line(self) -> i16 {
335        match self {
336            Self::Ntsc => 261,
337            Self::Pal | Self::Dendy => 311,
338        }
339    }
340
341    /// Last visible scanline (always 239).
342    #[must_use]
343    pub const fn last_visible_line(self) -> i16 {
344        239
345    }
346
347    /// Scanline at which V-blank starts (and `PPUSTATUS.VBLANK` is set on dot 1).
348    #[must_use]
349    pub const fn vblank_start_line(self) -> i16 {
350        match self {
351            Self::Ntsc | Self::Pal => 241,
352            Self::Dendy => 291,
353        }
354    }
355
356    /// Number of CPU cycles `$2000`/`$2001`/`$2005`/`$2006` writes are
357    /// ignored after a power-on / reset. Per nesdev wiki:
358    /// NTSC ≈ 29,658; PAL ≈ 33,132.
359    #[must_use]
360    pub const fn post_reset_mask_cycles(self) -> u32 {
361        match self {
362            Self::Ntsc => 29_658,
363            Self::Pal | Self::Dendy => 33_132,
364        }
365    }
366}
367
368/// v2.1.7 P5 — selectable 2C02 die revision, gating revision-dependent quirks.
369///
370/// Additive and **default-off**: the [`Default`] ([`Self::Rp2c02H`]) preserves
371/// `RustyNES`'s established behavior byte-for-byte, so `AccuracyCoin`, the
372/// commercial oracle, and the visual / audio regression suites are unaffected at
373/// the default. Only the opt-in [`Self::Rp2c02G`] selection changes any emulated
374/// behavior (see below).
375///
376/// Real RP2C02 dies shipped across several letter revisions. The one behavioral
377/// difference `RustyNES` currently models per-revision is the **OAMADDR
378/// (`$2003`) write-during-rendering OAM corruption** glitch: writing `$2003`
379/// while rendering is enabled on a visible / pre-render scanline copies one
380/// 8-byte OAM "row" from row 0 over the row the write's high bits target, on the
381/// next rendered dot (the same `CorruptOAM` mechanism the rendering-disable
382/// model uses; see `Ppu::process_oam_corruption`). A handful of titles —
383/// notably *Huge Insect* — trip it. It is **not** enabled on the default
384/// revision.
385///
386/// Honesty note (see `docs/accuracy-ledger.md`): the exact mapping of the
387/// `$2003` corruption onto specific 2C02 letter revisions is not firmly
388/// established in the public literature, and the precise per-title byte output
389/// of the glitch is not independently oracle-verified in this cut. `RustyNES`
390/// therefore offers the model as an opt-in approximation keyed to a single
391/// "earlier revision" selection ([`Self::Rp2c02G`]) rather than claiming exact
392/// silicon-revision fidelity. This is config, **not** save-state: like
393/// [`PpuRegion`] it is re-applied on load and is not part of the snapshot.
394#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash, Default)]
395pub enum PpuRevision {
396    /// Default. Later RP2C02 die (the "H"-class revision `RustyNES` has always
397    /// modeled). The OAMADDR (`$2003`) write-during-rendering OAM corruption is
398    /// **not** modeled, so the deterministic output is byte-identical to a build
399    /// without this feature.
400    #[default]
401    Rp2c02H,
402    /// Earlier RP2C02 die ("rev E+" in the nesdev notes). Additionally models the
403    /// OAMADDR (`$2003`) write-during-rendering OAM row-corruption glitch that
404    /// *Huge Insect* and a few other titles trip. Opt-in; changes emulated
405    /// behavior only for software that writes `$2003` mid-render.
406    Rp2c02G,
407}
408
409impl PpuRevision {
410    /// Whether this revision models the OAMADDR (`$2003`) write-during-rendering
411    /// OAM corruption glitch. Only [`Self::Rp2c02G`] does; the default returns
412    /// `false`, keeping the default build byte-identical.
413    #[must_use]
414    pub const fn models_oamaddr_corruption(self) -> bool {
415        matches!(self, Self::Rp2c02G)
416    }
417}
418
419/// v2.1.7 P5 — selectable power-up palette-RAM contents.
420///
421/// The 2C02's palette RAM is not cleared at power-on; different consoles (and
422/// thus different emulator authors' reference dumps) come up with different
423/// garbage. This is a documented power-up option, **default-off**: [`Default`]
424/// ([`Self::Zeroed`]) keeps `RustyNES`'s established all-zero power-up palette,
425/// so default rendering is byte-identical. It writes only `Ppu::palette_ram`,
426/// which is already part of the save-state snapshot, so it needs no
427/// snapshot-format change.
428#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash, Default)]
429pub enum PaletteInit {
430    /// Default. All 32 palette-RAM bytes power up to `0x00` — `RustyNES`'s
431    /// established deterministic power-up state. Byte-identical.
432    #[default]
433    Zeroed,
434    /// The canonical "Blargg" power-up palette dump (the 32-byte pattern used by
435    /// blargg's NES and mirrored by `TriCNES`'s `BlarggPalette`). A documented,
436    /// deterministic known pattern for software that samples uninitialized
437    /// palette RAM before writing it. Opt-in.
438    Blargg,
439}
440
441/// The canonical "Blargg" power-up palette-RAM contents (32 bytes). Mirrors
442/// `TriCNES`'s `BlarggPalette` table (`Emulator.cs`) verbatim. Only applied when
443/// [`PaletteInit::Blargg`] is selected.
444const BLARGG_POWER_UP_PALETTE: [u8; 32] = [
445    0x09, 0x01, 0x00, 0x01, 0x00, 0x02, 0x02, 0x0D, 0x08, 0x10, 0x08, 0x24, 0x00, 0x00, 0x04, 0x2C,
446    0x09, 0x01, 0x34, 0x03, 0x00, 0x04, 0x00, 0x14, 0x08, 0x3A, 0x00, 0x02, 0x00, 0x20, 0x2C, 0x08,
447];
448
449/// 2C02 PPU.
450///
451/// `tick(bus)` advances one PPU dot. The PPU is the master clock;
452/// `rustynes-core` calls it three times per CPU cycle (NTSC).
453#[derive(Debug)]
454#[allow(clippy::struct_excessive_bools)] // PPU's many 1-bit latches are spec
455pub struct Ppu {
456    /// Region (governs frame structure).
457    pub(crate) region: PpuRegion,
458
459    // === CPU-facing register state ===
460    pub(crate) ctrl: PpuCtrl,
461    pub(crate) mask: PpuMask,
462    /// Two-stage delay pipeline of `mask` consumed exclusively by the
463    /// pre-render dot-339 odd-frame skip check. `mask_for_skip_check` is
464    /// the value seen *this* dot; `mask_skip_pipe1` is the staged value
465    /// that will become visible *next* dot. Both shift at the end of every
466    /// `advance_dot`. The total visible delay between a PPUMASK write and
467    /// the dot-skip detector is two PPU clocks — enough to compensate for
468    /// the lockstep bus model applying `cpu_write` at the *start* of a CPU
469    /// cycle (before the cycle's 3 PPU ticks) while real hardware latches
470    /// the write at φ2 (effectively the *end* of the cycle). Required by
471    /// blargg `ppu_vbl_nmi/10-even_odd_timing`; tests 1-9 of the same
472    /// corpus are unaffected because rendering is enabled long before
473    /// the boundary.
474    pub(crate) mask_for_skip_check: PpuMask,
475    pub(crate) mask_skip_pipe1: PpuMask,
476    /// True from the odd-frame skip until the dot it lands on has run:
477    /// scanline 0's dot 0, which the skip REPLACED rather than reached. The
478    /// `NESdev` PPU rendering page: the skip jumps "(339,261) to (0,0),
479    /// replacing the idle tick at the beginning of the first visible scanline
480    /// with the last tick of the last dummy nametable fetch". That tick
481    /// belongs to the dummy fetch, so it drives a nametable address, and the
482    /// dot-0 rule in `tick` (a visible line's dot 0 drives the background CHR
483    /// address) must not apply to it (T-MMC3-BG-A12).
484    ///
485    /// Snapshotted (`PPU_SNAPSHOT_VERSION` 12): a state saved between the skip
486    /// and that dot would otherwise restore into a dot 0 that raises A12 the
487    /// uninterrupted run did not.
488    pub(crate) dot0_replaced: bool,
489    pub(crate) status: PpuStatus,
490    /// `$2003` OAMADDR.
491    pub(crate) oam_addr: u8,
492    /// `$2007` PPUDATA read buffer.
493    pub(crate) data_buffer: u8,
494    /// `mc-ppu-2007-render-buffer`: the most recent VRAM data-bus value (every
495    /// rendering fetch updates it). During rendering a `$2007` read returns THIS
496    /// (the pipeline's current fetch byte), not a read at `v` — the model the
497    /// `AccuracyCoin` `$2007 Stress` test brackets (`$2007` read on every dot).
498    pub(crate) render_data_bus: u8,
499    /// `mc-ppu-2007-render-buffer`: PPU-dot countdown from a `$2007` read during
500    /// rendering to the `PPUDATA` state machine's `data_buffer` reload. `TriCNES`
501    /// reloads `PPU_ReadBuffer` via a latch cascade ~4 dots after read-END
502    /// (`Emulator.cs` `PPU_DATA_StateMachine`); with R1's fixed CPU<->PPU mod-4
503    /// phase the hardware landing dot is a constant +5 dots from the
504    /// register-read sample point (empirical W2 sweep winner: 170/170 stable
505    /// reads). Set at the read (`RENDER_BUFFER_DOT_DELAY`),
506    /// decremented once per dot in `Ppu::tick` (end of the fetch dispatch); at
507    /// 0 the buffer latches `render_data_bus` (the fetch cadence's latched bus
508    /// value — never a fresh VRAM read, so zero new A12/mapper events).
509    /// 0 = inactive. Serialized in the PPU snapshot v3 tail (W3-Stage-4,
510    /// 2026-06-10) so an in-flight reload survives a save-state restore.
511    pub(crate) ppudata_sm_countdown: u8,
512    /// `mc-ppu-2007-render-buffer`: the v-glitch increment of the in-flight
513    /// `$2007` read is still pending — performed at the `TStep` (= the countdown
514    /// landing dot, after the buffer reload), per `TriCNES`
515    /// `PPU_DATA_StateMachine_Half`. Serialized in the PPU snapshot v3 tail
516    /// (paired with `ppudata_sm_countdown`).
517    pub(crate) ppudata_v_inc_pending: bool,
518    /// `mc-ppu-2007-render-buffer`: raw (pre-h-flip) sprite pattern bytes
519    /// captured by `fetch_sprite_tile` per slot, so the per-dot sprite-fetch
520    /// read cadence (dots 257-320) feeds `render_data_bus` the PT-lo / PT-hi
521    /// values the PPU drove on the bus (the `$2007` buffer captures the raw
522    /// bus byte, not the flipped shifter contents). Serialized in the PPU
523    /// snapshot v3 tail.
524    pub(crate) spr_fetch_lo_raw: [u8; 8],
525    /// See [`Self::spr_fetch_lo_raw`].
526    pub(crate) spr_fetch_hi_raw: [u8; 8],
527    /// `mc-ppu-2007-render-buffer`: the nametable address latched by the ALE
528    /// of sprite slot 0's first garbage NT read (dot 257) — captured BEFORE
529    /// the dot-257 copy-hori, so the read at dot 258 uses the OLD `v` (the
530    /// only sprite-interval read that does). Serialized in the PPU snapshot
531    /// v3 tail (refreshed every rendered scanline).
532    pub(crate) ppudata_spr0_nt_addr: u16,
533
534    // === v2.0.3 (ADR 0030) octal-latch / 2-cycle-ALE multiplexed-bus model ===
535    // Now the ONLY PPU fetch path (promoted from the experimental flag in
536    // v2.0.3; the superseded v2.0.2 whole-dot stand-in was retired). In continuous
537    // play these fields self-heal within a scanline (they reload on the next fetch
538    // ALE), but a mid-render rollback/save-state checkpoint can capture them live —
539    // so they ARE serialized in the `PPU_SNAPSHOT_VERSION` v5 tail (added in v2.0.3)
540    // for netplay-rollback determinism. That bump is ADDITIVE (pre-v5 blobs upconvert
541    // to the inactive rest defaults), NOT an ADR-0028 save-state format-epoch break.
542    //
543    // Ported from TriCNES (`TriCNES/Emulator.cs`, MIT, commit 9199870),
544    // the AccuracyCoin author's own cycle-accurate C# emulator (a detailed
545    // sub-cycle CPU/PPU/APU/DMA state machine), which is the
546    // ground-truth oracle for the "ALE + Read" / "Hybrid Addresses" tests (the
547    // vendored Mesen2 build does NOT pass them — see the ADR 0030 campaign audit).
548    //
549    // The PPU multiplexes its low 8 VRAM address pins (PA0-7) with the 8 data
550    // pins (AD7-0). A 74LS373-class external octal latch (`octal_latch`) captures
551    // A7-A0 on the address-latch-enable (ALE) half of each 2-cycle VRAM access;
552    // on the read half the PPU drives only A13-A8 (the fetch address's high 6
553    // bits, `& 0x3F00`) and the latch supplies A7-A0. The *effective* read
554    // address is therefore the splice `(fetch_addr & 0x3F00) | octal_latch`
555    // (TriCNES `FetchPPU`:153). When those halves desync — a mid-fetch `$2006`
556    // high-byte update, or a `$2007`-read ALE that overlaps the fetch cadence and
557    // freezes the latch on a stale DATA byte — the PPU reads a "hybrid" address it
558    // never coherently drove.
559    /// 74LS373 low-address octal latch (A7-A0). Loaded on each fetch's ALE
560    /// with the driven address's low byte; frozen (goes stale) across a
561    /// `$2007`-read/background-fetch ALE overlap. Serialized in the PPU snapshot
562    /// v5 tail (v2.0.3): it self-heals within a scanline in continuous play, but a
563    /// mid-render rollback checkpoint needs it restored for determinism.
564    pub(crate) octal_latch: u8,
565    /// The multiplexed PPU address/data bus. On a fetch's ALE (even) dot the full
566    /// driven 14-bit address is written here; on the read (odd) dot the DATA byte
567    /// is written back into the low 8 bits (the AD7-0 pins), so the register always
568    /// reflects the true multiplexed bus value. The effective read address is the
569    /// splice `(address_bus & 0x3F00) | octal_latch`. Serialized in the PPU snapshot
570    /// v5 tail (v2.0.3) for mid-render rollback determinism (self-heals on the next
571    /// fetch ALE in continuous play).
572    pub(crate) address_bus: u16,
573    /// Set on a fetch's ALE (even) dot after the address is driven onto
574    /// [`Self::address_bus`]; consumed (cleared) by the matching read (odd) dot's
575    /// [`Self::ale_splice`]. Distinguishes a read that followed a real ALE (main
576    /// background dispatch, phases 0/2/4/6 → 1/3/5/7) from a read with NO preceding
577    /// ALE (the dot-337-340 garbage nametable fetches), so the latter stays
578    /// behavior-neutral (drives + latches coherently in-place rather than splicing
579    /// a stale bus).
580    pub(crate) ale_armed: bool,
581    /// One-shot: the `$2007`-read ALE overlap froze `octal_latch` on the read's
582    /// DATA byte, so the next background PATTERN fetch reads
583    /// `(PAR-high 6):(stale low 8)` (`$0F03` -> `$0FFF`) — the "ALE + Read"
584    /// corruption. Consumed by the next pattern (BG-lo/BG-hi) fetch: the latch is
585    /// frozen at the PPUDATA state-machine landing dot and carried to the next
586    /// pattern read via the natural ALE/read multiplex.
587    pub(crate) pattern_latch_stale: bool,
588    /// The delayed-`CopyV` countdown (`TriCNES`
589    /// `PPU_Update2006Delay`, `Emulator.cs:1684-1704`). A `$2006` second write
590    /// that lands DURING rendering does NOT copy `t -> v` immediately; it stages
591    /// this PPU-dot countdown instead. While it runs, the background fetch
592    /// cadence keeps advancing coarse-X (via `inc_hori_v` at phase 7) and the
593    /// per-group phase-0 nametable ALE keeps loading `octal_latch` with the
594    /// CURRENT (pre-copy) `v`'s NT-low byte — so by the time the countdown lands
595    /// the latch NATURALLY holds the one-tile-ahead low byte (`$19`), and the
596    /// landing's `address_bus = v` splice yields the hybrid address (`$2F19`)
597    /// with no `+1 coarse-X` reconstruction. 0 = inactive.
598    pub(crate) copy_v_delay: u8,
599
600    // === Internal scroll/address registers (loopy v/t/x/w) ===
601    /// 15-bit "current VRAM address".
602    pub(crate) v: u16,
603    /// 15-bit "temporary VRAM address" (latched scroll/PPUADDR target).
604    pub(crate) t: u16,
605    /// 3-bit fine X scroll.
606    pub(crate) x: u8,
607    /// 1-bit write toggle for `$2005` / `$2006`.
608    pub(crate) w: bool,
609
610    // === Memory ===
611    /// Console-side nametable VRAM (CIRAM, 2 KiB). Owned by the PPU; the
612    /// mapper exposes a per-cart `nametable_address` mirroring map via
613    /// [`PpuBus::nametable_address`] so the PPU can read/write CIRAM directly
614    /// without going through `bus.ppu_read/write` for `$2000-$3EFF`.
615    pub(crate) ciram: Box<[u8]>,
616    /// Object Attribute Memory: 64 sprites × 4 bytes.
617    pub(crate) oam: Box<[u8]>,
618    /// Secondary OAM: up to 8 sprites for the next scanline. Populated
619    /// during sprite evaluation in Sprint 2-3.
620    pub(crate) secondary_oam: [u8; 32],
621    /// Palette RAM: 32 entries, 6-bit each (high 2 bits open-bus on read).
622    pub(crate) palette_ram: [u8; 32],
623
624    // === v2.1.4 F2.3 optional OAM decay (opt-in, default-OFF) ===
625    /// Per-8-byte-row last-touch timestamp, in **CPU cycles** (`dot_counter / 3`).
626    /// OAM is 256 bytes = 32 rows of 8 bytes; `oam_decay_cycles[addr >> 3]` is the
627    /// CPU cycle at which row `addr >> 3` was last refreshed by an OAM read/write.
628    /// Only consulted when [`Self::oam_decay_enabled`] is set AND the region is
629    /// NTSC/Dendy; otherwise it is dead state that never influences a read. Mirrors
630    /// Mesen2's `_oamDecayCycles` (`Core/NES/NesPpu.cpp`).
631    ///
632    /// Serialized as a *relative age* in the PPU snapshot v7 tail (see
633    /// `snapshot.rs`): the absolute timestamps reference the free-running,
634    /// **un-serialized** `dot_counter`, so a raw absolute value would be meaningless
635    /// after a rollback/restore rebased that counter. Storing `now - timestamp`
636    /// (and reconstructing `now - age` on load, relative to the live counter) keeps
637    /// a run-ahead / netplay `snapshot`→`restore` byte-identical to the forward run.
638    ///
639    /// Field POSITION here is not performance-relevant, and this was measured
640    /// rather than assumed (v2.3.1 G2, `docs/performance.md`): neither adding
641    /// `#[repr(C)]` nor moving this 256-byte cold array to the end of the struct
642    /// produced a reproducible change on any workload. `Ppu` is ~2.8 KB and stays
643    /// L1-resident across a frame, so layout has little left to buy.
644    pub(crate) oam_decay_cycles: [u64; 32],
645    /// Master enable for the OAM-decay model. **`false` by default** — a frontend /
646    /// config knob (re-applied on load like `region` / `active_palette`), NOT part
647    /// of the save-state. While `false`, every decay hook early-returns, so OAM
648    /// reads/writes touch neither this flag's siblings nor `oam_decay_cycles`, and
649    /// the framebuffer/audio/replay output is byte-identical to a decay-free build.
650    pub(crate) oam_decay_enabled: bool,
651
652    // === Open-bus latch (for $2000-$3FFF) ===
653    /// Most recent value driven onto the PPU bus by any register access.
654    pub(crate) open_bus: u8,
655    /// Per-bit-group decay counters (in CPU cycles) until each bit group of
656    /// the open-bus latch reads as 0.  Three groups, each with its own timer:
657    ///   `[0]` bits 0-4, `[1]` bit 5, `[2]` bits 6-7.
658    /// Required by `ppu_open_bus.nes` tests 7 and 9, which assert that some
659    /// reads refresh only a subset of the bit groups (e.g., reading $2002
660    /// must not refresh the low 5 bits' decay timer; palette $2007 reads
661    /// must not refresh the high 2 bits' decay timer).
662    pub(crate) open_bus_decay: [u32; 3],
663
664    // === NMI line + frame counter ===
665    /// `true` while the PPU is asserting NMI.
666    pub(crate) nmi_line: bool,
667    /// True for one frame after a `cpu_read_register($2002)` race so we
668    /// suppress the VBL flag set + NMI for that frame (per
669    /// `ppu_vbl_nmi/06-suppression.nes`). Toggled on the cycle the read
670    /// hits at scanline 241 dot 0 / dot 1.
671    pub(crate) suppress_vbl_this_frame: bool,
672    /// Last-observed A12 level, for edge-triggered notifications.
673    pub(crate) last_a12_level: bool,
674
675    // === Scanline FSM ===
676    /// Current dot (0..=340).
677    pub(crate) dot: u16,
678    /// Current scanline (-1 in pre-render, 0..=239 visible, 240 post-render,
679    /// 241..=260/310 vblank). Stored as i16 to allow temporary -1.
680    pub(crate) scanline: i16,
681    /// Frame counter (for odd-frame skip).
682    pub(crate) frame: u64,
683    /// `frame_complete` latch — set to `true` on the dot the PPU finishes
684    /// a frame; consumed by the run loop and cleared on next read.
685    pub(crate) frame_complete: bool,
686
687    // === Power-on / reset masking window ===
688    /// CPU cycles remaining in the post-reset masking window. While > 0,
689    /// writes to PPUCTRL/PPUMASK/PPUSCROLL/PPUADDR are silently ignored
690    /// (reads still work).
691    pub(crate) post_reset_mask_remaining: u32,
692
693    // === Background fetch + shift register state ===
694    /// Latched nametable byte from the current 8-cycle fetch group.
695    pub(crate) nt_latch: u8,
696    /// Latched attribute byte (palette) from the current 8-cycle fetch group.
697    pub(crate) at_latch: u8,
698    /// Latched BG pattern low byte from the current 8-cycle fetch group.
699    pub(crate) bg_lo_latch: u8,
700    /// Latched BG pattern high byte from the current 8-cycle fetch group.
701    pub(crate) bg_hi_latch: u8,
702    /// 16-bit BG pattern low shift register.
703    pub(crate) bg_shift_lo: u16,
704    /// 16-bit BG pattern high shift register.
705    pub(crate) bg_shift_hi: u16,
706    /// 16-bit attribute low shift register.
707    ///
708    /// Mirrors the 16-bit BG pattern shifters exactly: at each 8-dot
709    /// reload the latched attribute bit is expanded to a full byte
710    /// (`0x00` or `0xFF`) into bits 0-7, shifted left by 1 after each
711    /// emit, and shifted left by 8 at the pre-fetch boundary (dots 328 /
712    /// 336). Keeping it 16-bit (not the prior 8-bit + 1-bit-feed model)
713    /// is what keeps the attribute in lockstep with the pattern bits
714    /// through the dots 321-336 pre-fetch region, where `shift_bg` does
715    /// not run and only the explicit `<<= 8` advances the registers.
716    pub(crate) at_shift_lo: u16,
717    /// 16-bit attribute high shift register. See [`Self::at_shift_lo`].
718    pub(crate) at_shift_hi: u16,
719    /// Optional per-tile extended attribute (MMC5 `ExGrafix`). Latched at
720    /// the NT-byte fetch boundary; consumed by AT / BG-low / BG-high
721    /// fetches in the same 8-dot group.
722    pub(crate) ex_attr_latch: Option<ExAttribute>,
723    /// Optional vertical split-screen state (MMC5 `$5200`-`$5202`). Latched
724    /// at the NT-byte fetch boundary; consumed by AT / BG-low / BG-high
725    /// fetches in the same 8-dot group. When `Some`, the BG fetches use the
726    /// alt region's nametable address, attribute address, fine-Y, and CHR
727    /// bank instead of the values derived from `v`.
728    pub(crate) bg_split_latch: Option<BgSplitState>,
729
730    // === Sprite rendering state ===
731    /// Per-sprite shift registers (low + high pattern).
732    pub(crate) spr_shift_lo: [u8; 8],
733    pub(crate) spr_shift_hi: [u8; 8],
734    /// Per-sprite latched attribute byte.
735    pub(crate) spr_attr: [u8; 8],
736    /// Per-sprite X-coordinate counter.
737    pub(crate) spr_x: [u8; 8],
738    /// v2.0 (ppu-sprite-shifter-counter): per-sprite persistent "halted" latch.
739    /// Set when the X-counter reaches 0 (the sprite is drawing) and PERSISTS
740    /// across a render-disable and the frame boundary; re-armed to "counting"
741    /// at dot 339 for the loaded slots. Carries the Stale Sprite Shift Regs
742    /// t5/6 behavior (a reloaded-but-halted sprite draws on the next rendering
743    /// re-enable). Default build: absent (legacy `spr_x == 0` predicate).
744    pub(crate) spr_halted: [bool; 8],
745    /// Number of sprites loaded for the current scanline.
746    pub(crate) spr_count: u8,
747    /// `true` if sprite 0 is in the current scanline's sprite line-up.
748    pub(crate) spr_zero_in_line: bool,
749
750    // === Per-dot sprite-evaluation FSM state ===
751    /// Sprite-eval read latch: byte read from primary OAM on odd cycles
752    /// (1, 3, 5, ...) of dots 65-256, consumed by the immediately-following
753    /// even-cycle write into secondary OAM.
754    pub(crate) sprite_eval_read_latch: u8,
755    /// Primary-OAM sprite index 0..=63 walked during dots 65-256.
756    pub(crate) sprite_eval_n: u8,
757    /// Per-sprite byte index 0..=3 walked during dots 65-256 (drives the
758    /// buggy `n+m` increment when overflow detection mode is active).
759    pub(crate) sprite_eval_m: u8,
760    /// Number of in-range sprites found so far in this scanline's eval pass.
761    pub(crate) sprite_eval_found: u8,
762    /// Write index into `secondary_oam` (0..=31). Tracks how many bytes the
763    /// per-dot FSM has committed so far.
764    pub(crate) sprite_eval_sec_idx: u8,
765    /// `true` when the current sprite (the one whose `y` byte just tested
766    /// in-range) is still being copied — bytes 1, 2, 3 land in subsequent
767    /// even-dot writes.
768    pub(crate) sprite_eval_copying: bool,
769    /// `true` when eval has exhausted primary OAM (n wrapped past 63) or
770    /// overflow has been detected — remaining dots 65-256 idle out.
771    pub(crate) sprite_eval_done: bool,
772    /// `true` when 8 in-range sprites have been latched and the FSM is
773    /// in overflow-detection mode (buggy `n+m` increment active).
774    pub(crate) sprite_eval_overflow_search: bool,
775    /// Eval-side latch for "sprite 0 is in the line being evaluated."
776    /// Set during the current scanline's eval pass (dots 65..=256) when
777    /// sprite 0 lands in-range; committed to [`Self::spr_zero_in_line`]
778    /// at dot 256 alongside [`Self::spr_count`]. Keeping the eval-side
779    /// latch separate from the rendering-side flag ensures the FSM
780    /// doesn't trample the CURRENT scanline's sprite-0-hit signal while
781    /// it's still being read by the dots 1..=256 sprite-pixel evaluator.
782    pub(crate) sprite_eval_zero_found: bool,
783    /// Phase 3a flag — tracks whether current scanline's eval is on
784    /// its FIRST iteration (PPU cycle 66, first y-test).  Set at
785    /// dot 0 of each visible scanline; cleared after the first y-test
786    /// fires (in-range or not).  Per Mesen2 `ProcessSpriteEvaluation`
787    /// line 1040-1044, sprite-zero fires IFF the FIRST y-test is in
788    /// range — not "first in-range sprite found".  When OAMADDR is 0
789    /// at eval start and OAM[0].y is in range, this matches the legacy
790    /// `n == 0` check.  When OAMADDR != 0, this fires on whichever
791    /// sprite the start position points to (sprite at OAMADDR / 4)
792    /// if its y is in range, else NO sprite-zero is detected.
793    pub(crate) sprite_eval_first_iter: bool,
794
795    /// v2.0 Tier 1.2 — isolated OAM-data-bus model of the `NESdev`-documented PPU
796    /// sprite-evaluation datapath (`NESdev` wiki "PPU sprite evaluation"). These
797    /// fields exist ONLY under `ppu-oam-data-bus` and are read solely by `$2004`
798    /// during rendering — the rendering / sprite-zero / overflow / MMC3
799    /// sprite-fetch FSM uses `secondary_oam` + `sprite_eval_*` + `spr_*`, all
800    /// untouched. `oam_bus_copybuffer` is the value `$2004` returns while the
801    /// screen is drawn (the byte currently on the OAM data bus).
802    ///
803    /// Provenance: the OAM-data-bus and sprite-evaluation model is **derived
804    /// from Mesen2's `NesPpu.cpp`** (`ProcessSpriteEvaluation` / `ReadSpriteRam`),
805    /// GPL-3.0-or-later. See NOTICE and docs/originality-and-provenance.md (Section 1).
806    pub(crate) oam_bus_copybuffer: u8,
807    /// Parallel secondary OAM (the 32-byte sprite line buffer) for the bus model only.
808    pub(crate) oam_bus_secondary: [u8; 32],
809    /// Eval-pointer sprite index (0..=63) — which of the 64 primary sprites is examined.
810    pub(crate) oam_bus_addr_h: u8,
811    /// Eval-pointer byte-in-sprite (0..=3) — Y / tile / attr / X.
812    pub(crate) oam_bus_addr_l: u8,
813    /// Write index into the parallel secondary OAM.
814    pub(crate) oam_bus_secondary_addr: u8,
815    /// Primary OAM fully scanned / wrapped for this scanline.
816    pub(crate) oam_bus_copy_done: bool,
817    /// Currently copying an in-range sprite.
818    pub(crate) oam_bus_sprite_in_range: bool,
819    /// The 8-sprite-overflow PPU-bug countdown.
820    pub(crate) oam_bus_overflow_counter: u8,
821
822    /// OAM-corruption model — faithful port of `TriCNES`'s eval-pointer
823    /// machinery (`Emulator.cs` `PPU_Render_SpriteEvaluation` lines
824    /// 2664-2770 + `CorruptOAM` lines 2635-2651). Replaces the earlier
825    /// Mesen2 `_corruptOamRow` row-flag model (`dot >> 1` index), which
826    /// Mesen ships OFF by default (`EnablePpuOamRowCorruption=false`) and
827    /// documents as unfinished. The bug it fixes: SMB3 (MMC3) toggles
828    /// PPUMASK mid-visible-scanline to split its HUD; NMI/DMA jitter
829    /// shifts the disable dot, and the raw-dot row index intermittently
830    /// landed on Mario's OAM row (offset 40), wiping his sprite.
831    ///
832    /// `TriCNES` model: when rendering is disabled (1 -> 0) DURING sprite
833    /// evaluation (dots 1-64, secondary-OAM clear, NOT the pre-render
834    /// line), the corruption is DEFERRED — `oam_corruption_pending` is
835    /// set and `oam_corruption_index` captures the live secondary-OAM
836    /// write pointer (`OAM2Address`) at that instant. When rendering
837    /// RE-ENABLES (or at the pre-render line), one OAM "row" of 8 bytes
838    /// is replaced from row 0: `oam[index*8 + i] = oam[i]` for i in
839    /// 0..8 (index 0x20 wraps to 0), and `secondary_oam[index] =
840    /// secondary_oam[0]`.
841    ///
842    /// `oam2_addr` is the `OAM2Address` analogue maintained across the
843    /// dots 1-64 clear window (our dots 65-256 active eval already walks
844    /// `sprite_eval_sec_idx`, but the SMB3 HUD-split disable lands in the
845    /// clear window where `sprite_eval_sec_idx` is held at 0, so the
846    /// dedicated pointer is required to capture the right index).
847    ///
848    /// `oam_corruption_disabled` / `_instant` mirror `TriCNES`'s
849    /// `PPU_OAMCorruptionRenderingDisabledOutOfVBlank` (1-dot-delayed,
850    /// armed by the `$2001` write-delay) and `..._Instant` (the
851    /// data-bus-immediate path: OAM eval observes the disable the same
852    /// cycle). The disable edge is captured into `pending`/`index`
853    /// during the dots 1-64 eval window; the actual corruption is
854    /// committed at re-enable / pre-render.
855    ///
856    /// None of these fields are persisted in the PPU snapshot — like the
857    /// rest of the per-dot sprite-eval FSM state (`sprite_eval_*`), they
858    /// re-derive within a scanline/frame, matching the prior row-flag
859    /// OAM-corruption model (also un-snapshotted).
860    pub(crate) oam_corruption_pending: bool,
861    pub(crate) oam_corruption_index: u8,
862    pub(crate) oam_corruption_disabled: bool,
863    pub(crate) oam_corruption_disabled_instant: bool,
864    /// `OAM2Address` analogue: the secondary-OAM write pointer as it
865    /// walks the dots 1-64 secondary-OAM clear window. Reset to 0 at the
866    /// dot-1 boundary and incremented once per even clear dot, masked to
867    /// 0x1F, exactly as `TriCNES` drives `OAM2Address` during dots 1-64.
868    pub(crate) oam2_addr: u8,
869    /// `OAM2Address` as a LIVE counter across sprite fetch (dots 257-320),
870    /// and the "OAM2 Overflowed" flag that freezes it.
871    ///
872    /// Both rules are stated outright by `AccuracyCoin`'s own source, which is
873    /// the specification here (MIT; stimulus, not a reference implementation):
874    ///
875    /// > When OAM2 is full, the PPU prevents further increments of the OAM2
876    /// > Address, so the OAM2 Address is frozen at index 0. This flag ... is
877    /// > cleared if rendering is enabled during dots 63, 255, and 339.
878    ///
879    /// The counter advances on EVEN dots (32 bytes across the 64-dot fetch).
880    /// That yields 33 candidate increments over dots 256..=320, and the flag
881    /// is what reconciles it: the 32nd wraps 0x1F -> 0 AND raises the flag,
882    /// which suppresses the 33rd, leaving the address resting at 0. So the
883    /// documented "`$2004` during dots 321-340 reads OAM2[0]" is not a special
884    /// case — it is the ordinary end state of the counter, and the flag is
885    /// load-bearing for it. Both `AccuracyCoin` `Misaligned OAM2 Address` and
886    /// `Frozen OAM2 Increment` are explained by this one mechanism.
887    ///
888    /// **Snapshotted, in the `PPU_SNAPSHOT_VERSION` 9 tail** — unlike
889    /// `oam2_addr` and the rest of the per-dot sprite-eval FSM, which re-derive
890    /// within a scanline. This one cannot: the whole point of the live counter
891    /// is that it REMEMBERS increments it did not take while rendering was off,
892    /// so a restore that re-derived it would discard exactly the state the
893    /// counter exists to carry. A pre-v9 blob has no such field and restores to
894    /// `0` / `false` / `false`, which is the power-on state.
895    pub(crate) oam2_fetch_addr: u8,
896    pub(crate) oam2_overflowed: bool,
897    /// The freeze flag LATCHED at the start of sprite fetch (dot 257).
898    ///
899    /// Read instead of `oam2_overflowed` by the sprite loader, because in the
900    /// ordinary case the counter wraps and raises the flag near the END of the
901    /// fetch window (dot 318) -- reading the live flag would then retroactively
902    /// force the last slot to OAM2[0] and corrupt every game's final sprite.
903    /// The hardware behaviour under test is about the flag's state going IN.
904    pub(crate) oam2_fetch_frozen: bool,
905    /// Previous-tick rendering-enabled state — tracks the rising /
906    /// falling edge of `mask.rendering_enabled()` so the 1->0 edge
907    /// BG-shifter fix-up fires on the correct transition.
908    pub(crate) prev_rendering_enabled: bool,
909    /// v2.0 (ported from branch `ae30785`) — 1-PPU-dot-delayed rendering-enabled
910    /// gate. Per Mesen2 `NesPpu::UpdateState`: a `$2001` write toggling
911    /// `SHOW_BG|SHOW_SPRITE` takes effect on the rendering pipeline one PPU dot
912    /// later (the `mask` bit-fields update immediately for pixel output; this
913    /// delayed copy gates the fetch/shift/sprite-eval pipeline). When rendering
914    /// is stable this equals `mask.rendering_enabled()`, so only mid-scanline
915    /// `$2001` toggles observe the delay. The `ppu-sprite-shifter-counter`
916    /// feature reads it (via `rendering_gate`); the default build uses the
917    /// immediate value, so flag-off is byte-identical. Updated at tick end.
918    pub(crate) rendering_enabled_delayed: bool,
919
920    /// Two-dots-ago rendering value, for the v2.6.18 depth knob only.
921    #[cfg(feature = "phi2-write-sweep")]
922    pub(crate) render_gate_prev2: bool,
923
924    /// Rendering as it stood TWO dots ago -- the gate the dot-256 vertical
925    /// increment reads, and the only consumer that needs a second stage.
926    ///
927    /// v2.6.18. `AccuracyCoin`'s `Frozen OAM2 Increment` test 2 states the rule
928    /// this models: *"Rendering is enabled on dot 256, but the PPU's vertical
929    /// scroll is NOT incremented."* That sentence only parses if the enable IS
930    /// on dot 256 -- so an enable in force from the start of dot 256 must not
931    /// fire dot 256's increment, and the increment therefore sees the mask one
932    /// dot deeper than [`Ppu::rendering_enabled_delayed`] does.
933    ///
934    /// Measured rather than fitted: a `$2001` write whose effect lands during
935    /// dot N is in force from the start of N+1, and the ROM over-determines the
936    /// depth (the dot-256 increment must not fire, the dot-257 latch must), so
937    /// `d = 257 - 255 = 2`. Both neighbouring depths fail -- one dot shallower
938    /// is byte-identical to the unfixed build, one dot deeper breaks test 2.
939    pub(crate) rendering_enabled_delayed2: bool,
940
941    /// The dot-339 sprite re-arm, deferred past scanline 0's first pixel
942    /// because the odd-frame skip removed the dot at which hardware sees it.
943    ///
944    /// v2.9.5. The shifters are told to start counting on dot 339 and see the
945    /// signal a dot late, at 340. When an odd frame skips pre-render dot 340,
946    /// every loaded shifter therefore starts scanline 0 still in the drawing
947    /// state. It outputs its first pixel at X=0 and shifts, then counts one dot
948    /// late, so pixels 1-7 land where they always do and only the first moves
949    /// (`forums.nesdev.org/viewtopic.php?t=26291`, from `Visual2C02` analysis).
950    /// This is what `AccuracyCoin`'s `Sprites On Scanline 0` reads as a
951    /// composite 2C02: code 1, where the missing alternation had read as an
952    /// RGB PPU, code 2.
953    ///
954    /// Set by the skip in `advance_dot` and cleared after pixel 0 in
955    /// `emit_pixel`. It lives across the frame boundary, where run-ahead and
956    /// save states snapshot, so it is serialized (`PPU_SNAPSHOT_VERSION` 11).
957    pub(crate) spr_rearm_deferred: bool,
958
959    /// v2.6.18 (`phi2-write-sweep` only): a shared four-stage `$2001` history,
960    /// newest first, that each swept consumer gate reads at its OWN depth.
961    ///
962    /// The first version of `OAM2_GATE_LAG` borrowed the odd-frame-skip
963    /// pipeline, which capped the reachable depth at two AND coupled two
964    /// unrelated consumers. The measurement that mattered lives past that cap:
965    /// at the shipped placement `Frozen OAM2 Increment` closes only with the
966    /// OAM2 gate deeper than the skip pipeline can express.
967    #[cfg(feature = "phi2-write-sweep")]
968    pub(crate) sweep_mask_history: [PpuMask; 4],
969
970    /// v2.0 Phase 6 (`mc-ppu-subpos`): the analog `$2001` BG-shift-register
971    /// RELOAD delay. The shifter reload gates on `bg_reload_render`, which tracks
972    /// the live `self.mask` rendering-enable bit EXCEPT during the
973    /// `MASK_WRITE_DELAY`-dot window after a `$2001` write, where it stays frozen
974    /// at its prior value (`TriCNES` gates the fetch/reload on
975    /// `PPU_Mask_Show*_Delayed` while the per-half-dot SHIFT runs on the
976    /// IMMEDIATE mask). So on a render re-enable the shifter advances for
977    /// `MASK_WRITE_DELAY` dots (injecting the serial-in '1') BEFORE the reload
978    /// resumes -> one reload is SKIPPED and the accumulated '1's reach the output
979    /// (BG Serial In), without perturbing the sprite/pixel/shift path. Because it
980    /// re-syncs to the live mask whenever settled, a direct mask set (unit tests,
981    /// save-state restore) leaves it consistent. `mask_write_delay` is the
982    /// remaining freeze countdown (0 = settled). Serialized in the PPU
983    /// snapshot v3 tail (W3-Stage-4, 2026-06-10) so an in-flight freeze
984    /// survives a save-state restore.
985    pub(crate) bg_reload_render: bool,
986    pub(crate) mask_write_delay: u8,
987
988    /// v1.4.0 Workstream F (F1) — scanline-stable rendering-classification
989    /// cache. `visible` / `pre_render` / `render_line` are pure functions of
990    /// `self.scanline` + `self.region`, so they only change when the scanline
991    /// advances. The hot per-dot `tick` recomputes them ~7 branches deep
992    /// 89,342 times/frame; instead we recompute them once when the scanline
993    /// changes (detected via the `flags_cached_scanline` sentinel) and read the
994    /// cached copies on every other dot. Byte-identical by construction (same
995    /// values, computed less often) and self-healing across reset / save-state
996    /// restore (the sentinel starts mismatched, forcing a recompute on the
997    /// first tick). NOT part of the PPU snapshot — pure derived data.
998    pub(crate) cached_visible: bool,
999    pub(crate) cached_pre_render: bool,
1000    pub(crate) cached_render_line: bool,
1001    /// v2.2.3 P2 — `true` on an **idle** line: not visible, not pre-render, and
1002    /// not the VBL-set line (`vblank_start_line`). On NTSC that is line 240
1003    /// (post-render) plus lines 242..=260 — 20 of 262. Such a line issues no
1004    /// fetch, emits no pixel, runs no sprite evaluation, and raises no event, so
1005    /// its dots are eligible for [`Self::tick_idle_line_fast`]. Derived from
1006    /// `scanline` + `region` exactly like the three flags above, and keyed by
1007    /// the same [`Self::flags_cached_scanline`] sentinel.
1008    #[cfg(feature = "ppu-idle-line-fast")]
1009    pub(crate) cached_idle_line: bool,
1010    /// Sentinel: the scanline `cached_*` were last computed for. `i16::MIN`
1011    /// (an impossible scanline) forces a recompute on the first tick.
1012    pub(crate) flags_cached_scanline: i16,
1013
1014    /// Active output palette. Defaults to the 2C02 composite palette so normal
1015    /// NES/Famicom rendering is byte-for-byte unchanged; set to one of the RGB
1016    /// variants for Vs. System / PlayChoice-10 carts (see [`Ppu::set_palette`]).
1017    /// Construction-time configuration only — never mutated during emulation, so
1018    /// it is intentionally NOT part of the PPU save-state snapshot (it is
1019    /// re-derived from the cartridge header on load).
1020    pub(crate) active_palette: crate::palette::PpuPalette,
1021    /// v2.8.0 Phase 4 — precomputed `(emphasis bits << 6) | color` → RGBA8
1022    /// lookup (8 emphasis combinations × 64 colors = 512 entries, 2 KiB).
1023    /// Built from the same pure [`crate::palette::palette_color_to_rgba`]
1024    /// the per-pixel path used to call, so it is byte-identical by
1025    /// construction; rebuilt whenever [`Ppu::set_palette`] changes the
1026    /// active palette. Saves a palette-variant match + emphasis branches
1027    /// per emitted pixel (61,440/frame). NOT part of the save-state
1028    /// (derived data).
1029    pub(crate) rgba_lut: [[u8; 4]; 512],
1030    /// v1.1.0 beta.1 (T-110-A3) — optional custom 64-entry base palette from a
1031    /// loaded `.pal` file. `None` (default) = use the built-in palette for the
1032    /// active [`crate::palette::PpuPalette`], so default rendering is
1033    /// byte-identical. When `Some`, [`Self::rgba_lut`] is built from it via the
1034    /// composite emphasis model. A frontend presentation override — NOT part of
1035    /// the save-state (it persists across save/load like the active palette).
1036    pub(crate) custom_palette: Option<[[u8; 3]; 64]>,
1037    /// True when this is a 2C05 PPU: `$2000`/`$2001` are swapped and `$2002`
1038    /// returns the 2C05 sub-variant identifier in its low bits. Default false
1039    /// (a 2C02 / 2C03 / 2C04, none of which swap or report an id).
1040    pub(crate) is_2c05: bool,
1041    /// 2C05 sub-variant `$2002` identifier byte (e.g. `$3D` for 2C05-02). Only
1042    /// consulted when [`Self::is_2c05`] is true. Combined into the low 5 bits
1043    /// of a `$2002` read per nesdev "PPU registers" §2C05 identifier.
1044    pub(crate) id_2c05: u8,
1045
1046    /// v2.1.7 P5 — selected 2C02 die revision. Gates the OAMADDR (`$2003`)
1047    /// write-during-rendering OAM corruption glitch (see [`PpuRevision`]). The
1048    /// [`PpuRevision::default`] ([`PpuRevision::Rp2c02H`]) models NO extra
1049    /// corruption, so the default build is byte-identical. Construction / config
1050    /// only — never mutated by emulation and, like [`Self::active_palette`] /
1051    /// [`Self::region`], re-applied on load rather than serialized in the
1052    /// snapshot (the corruption *state* it can arm — `oam_corruption_pending` /
1053    /// `oam_corruption_index` — IS in the v6 snapshot tail, so an armed
1054    /// corruption still round-trips).
1055    pub(crate) die_revision: PpuRevision,
1056    /// v2.1.7 P5 — the power-up palette-RAM contents selected for this PPU (see
1057    /// [`PaletteInit`]). Stored so a power-cycle can re-apply it after the PPU is
1058    /// reconstructed. The [`PaletteInit::default`] ([`PaletteInit::Zeroed`])
1059    /// leaves palette RAM all-zero (the established default), keeping default
1060    /// rendering byte-identical. Config, not serialized (it writes
1061    /// [`Self::palette_ram`], which the snapshot already carries).
1062    pub(crate) power_up_palette: PaletteInit,
1063
1064    /// Framebuffer (RGBA8). Filled by Sprint 2-2/2-3 rendering.
1065    pub(crate) framebuffer: Box<[u8]>,
1066
1067    /// v1.1.0 beta.1 (T-110-A1) — parallel per-pixel **palette-index**
1068    /// framebuffer for the true composite `NES_NTSC` filter. Each entry is the
1069    /// 9-bit `(emphasis << 6) | colour_index` value (0..=511) written in the
1070    /// same emit path as [`Self::framebuffer`], so it is a faithful index-space
1071    /// mirror of the RGBA output. The frontend uploads it as an `R16Uint`
1072    /// texture and reconstructs the composite signal in a shader. Purely an
1073    /// output buffer: it changes no logical state, so the determinism /
1074    /// `AccuracyCoin` contract is unaffected. Unlike `framebuffer` (which IS in
1075    /// the save-state), this and `dot_counter` / `frame_ntsc_phase` are NOT
1076    /// serialized — they are regenerated on the next emitted frame, so a state
1077    /// loaded while paused shows correct NTSC from the first frame after resume.
1078    pub(crate) index_framebuffer: Box<[u16]>,
1079    /// How many dots took the specialized fast path (`ppu-fetch-trace` only).
1080    ///
1081    /// Not telemetry. It exists so a test can assert it EXERCISED the fast path
1082    /// rather than passing because the path was never entered — the failure this
1083    /// project keeps finding, where a check agrees about something it never
1084    /// reached. A review of #450 claimed the fast path bypasses the fetch trace;
1085    /// the test refuting that is worthless unless it can show the path ran, and
1086    /// this is how it shows it.
1087    #[cfg(feature = "ppu-fetch-trace")]
1088    pub fast_path_hits: u64,
1089
1090    /// Free-running PPU master-cycle counter (one increment per [`Self::tick`]),
1091    /// the basis for the per-frame NTSC colour phase. Output-only / cosmetic
1092    /// (drives only the optional NTSC filter's dot-crawl); not part of the
1093    /// save-state. Wraps harmlessly.
1094    pub(crate) dot_counter: u64,
1095
1096    /// NTSC composite colour phase snapshotted at each frame boundary, the
1097    /// per-frame `videoPhase` consumed by the `NES_NTSC` filter (the shader
1098    /// derives the per-scanline / per-pixel phase from this base). `0..=2` on
1099    /// NTSC; on PAL/Dendy (no 3-phase crawl) it is the frame parity (`0..=1`).
1100    /// Cosmetic; not part of the save-state.
1101    pub(crate) frame_ntsc_phase: u8,
1102
1103    /// v1.7.0 "Forge" Workstream F3 — PPU extra-scanlines overclock.
1104    ///
1105    /// Number of EXTRA blank scanlines to insert into the vblank period each
1106    /// frame (immediately before the pre-render line), at the existing dot
1107    /// resolution (Mesen2 `UpdateTimings`). These lines render nothing, emit no
1108    /// pixels, set/clear no PPU flags, and fire no VBL/NMI/A12 events — they are
1109    /// pure additional CPU run-time per frame, giving games more compute headroom
1110    /// without altering the visible image. **Off by default (`0`)**; the
1111    /// `advance_dot` insertion path is entirely guarded by `extra_scanlines != 0`,
1112    /// so at the default this field changes nothing and the frame is
1113    /// byte-identical to stock. Distinct from the CPU-multiplier overclock
1114    /// (`Nes::set_cpu_overclock`, v3.1.0, in the bus). A frontend config knob,
1115    /// NOT part of the save-state (re-applied by the frontend on restore, like
1116    /// `region` / `active_palette`).
1117    pub(crate) extra_scanlines: u16,
1118    /// v3.1.0 (`T-SPRITE-LIMIT`, FE-02): draw the sprites beyond the eighth on
1119    /// a scanline. **Render-only**: sprite evaluation, secondary OAM, the
1120    /// overflow flag, sprite-0 hit and every real sprite fetch (with its A12
1121    /// edges) are untouched. The extra sprites' patterns are read after the
1122    /// eight real fetches, through [`PpuBus::chr_reads_are_pure`] boards only,
1123    /// with no A12 notification, and they draw behind all eight hardware
1124    /// sprites (a higher OAM index is a lower priority). Off by default;
1125    /// configuration, carried across a power cycle by
1126    /// [`Self::adopt_settings_from`] and in movies / netplay by the core's
1127    /// `HardwareOptions`.
1128    pub(crate) sprite_limit_disabled: bool,
1129    /// v3.1.0 — the extra sprites fetched for the next scanline (snapshot v13):
1130    /// how many, and for each the h-flip-applied pattern bytes, attributes and
1131    /// X. Always 0 while `sprite_limit_disabled` is off.
1132    pub(crate) spr_extra_count: u8,
1133    pub(crate) spr_extra_lo: [u8; MAX_EXTRA_SPRITES],
1134    pub(crate) spr_extra_hi: [u8; MAX_EXTRA_SPRITES],
1135    pub(crate) spr_extra_attr: [u8; MAX_EXTRA_SPRITES],
1136    pub(crate) spr_extra_x: [u8; MAX_EXTRA_SPRITES],
1137    /// v1.7.0 F3 — countdown of extra blank scanlines remaining for the CURRENT
1138    /// frame's vblank insertion. Loaded from [`Self::extra_scanlines`] when the
1139    /// PPU reaches the insertion point and decremented one extra line at a time.
1140    /// `0` when no insertion is in flight. Snapshotted (snapshot v4) so a
1141    /// save-state taken mid-insertion restores the in-flight countdown rather
1142    /// than resuming as `0` and desyncing. The configured count itself
1143    /// (`extra_scanlines`) stays a non-persisted frontend knob, re-applied on
1144    /// restore. At the default `extra_scanlines == 0` this is always `0`.
1145    pub(crate) extra_lines_remaining: u16,
1146
1147    /// v2.1.8 A1 — enable the specialized straight-line per-dot fast path for
1148    /// the common visible-scanline BG-render window (see [`Self::tick`] and
1149    /// `docs/performance.md`). When `true`, visible scanline dots `1..=256`
1150    /// whose per-dot state is provably "undisturbed" (no pending `$2006`
1151    /// copy-V, no PPUMASK write-delay, no PPUDATA state machine in flight, no
1152    /// armed/pending OAM-corruption, warm scanline classification cache,
1153    /// stable rendering-enable) are dispatched to
1154    /// [`Self::tick_visible_render_fast`], which executes the identical
1155    /// helper sequence with the statically-dead event/bookkeeping branches
1156    /// pruned. Any disturbance drops instantly back to the exact path.
1157    /// **Not serialized** — a frontend/config knob re-applied on restore.
1158    ///
1159    /// **Default `true` since the v2.2.3 performance pass (was default-OFF for
1160    /// v2.1.8 .. v2.2.2).** A1 shipped this off deliberately: it was that
1161    /// roadmap's highest-risk item, and keeping it off left the shipped build
1162    /// byte-identical while the differential test and the oracle suites proved
1163    /// correctness. Both conditions A1 named for promotion are now met —
1164    ///
1165    /// * **byte-identity**, held continuously since v2.1.8 by
1166    ///   `crates/rustynes-test-harness/tests/fast_dotloop_diff.rs`, which runs
1167    ///   a ROM corpus through BOTH paths and asserts identical framebuffer,
1168    ///   palette-index framebuffer, audio, CPU-cycle count and full core
1169    ///   snapshot **every frame** (so the fast path has never been unproven —
1170    ///   only unshipped); and
1171    /// * **a clean-host Criterion confirmation** of the win: `full_frame`
1172    ///   `nes_run_frame_nestest` 4.4343 ms -> 3.9331 ms, **-11.3%**, on a quiet
1173    ///   host, reproducing A1's interleaved +12.3% measurement. The
1174    ///   rendering-disabled `flowing_palette` workload is unchanged (-0.07%,
1175    ///   noise) because its guard bails at `rendering_enabled()`.
1176    ///
1177    /// Promotion changes the *default*, not the behaviour: the frame the fast
1178    /// path produces is the frame the exact path produces, by construction and
1179    /// by test. `false` still selects the fully-general per-dot path and
1180    /// remains the fallback for any future doubt.
1181    pub(crate) fast_dotloop: bool,
1182
1183    /// Optional per-PPU-dot state trace (Session-10 observability
1184    /// tooling). Gated on the `ppu-state-trace` cargo feature so
1185    /// the default build pays no memory or codegen cost. See
1186    /// `docs/adr/0005-ppu-state-trace.md`.
1187    #[cfg(feature = "ppu-state-trace")]
1188    pub(crate) state_trace: Option<crate::state_trace::PpuStateTrace>,
1189
1190    /// The CPU cycle the dots being ticked belong to, stamped into every state
1191    /// record. Written once per CPU cycle by the bus, before those dots run.
1192    ///
1193    /// Feature-gated, so the default build carries neither the field nor the
1194    /// store: this is bookkeeping for a diagnostic, and the tick path is the
1195    /// hottest loop in the emulator.
1196    #[cfg(feature = "ppu-state-trace")]
1197    pub(crate) trace_cpu_cycle: u64,
1198    /// Per-dot PPU bus address capture. See [`crate::fetch_trace`].
1199    #[cfg(feature = "ppu-fetch-trace")]
1200    pub(crate) fetch_trace: Option<crate::fetch_trace::FetchTrace>,
1201
1202    /// v1.2.0 beta.2 (Workstream C3) — per-pixel HD-pack tile-source buffer
1203    /// (256 × 240 [`HdTileSource`] records), written in [`Self::emit_pixel`]
1204    /// in lockstep with [`Self::index_framebuffer`]. Output-only telemetry,
1205    /// gated on the `hd-pack` cargo feature so the default build pays no
1206    /// memory or codegen cost. Not part of the save-state.
1207    #[cfg(feature = "hd-pack")]
1208    pub(crate) hd_tile_source: Box<[HdTileSource]>,
1209
1210    /// v1.2.0 beta.2 (Workstream C3) — BG tile CHR base address latched at
1211    /// `fetch_bg_lo` time, then reloaded into the 2-stage `hd_bg_addr_*`
1212    /// queue in `reload_bg_shift_regs` so it tracks the BG pattern shift
1213    /// registers tile-for-tile. Pure telemetry; only touched when `hd-pack`
1214    /// is enabled.
1215    #[cfg(feature = "hd-pack")]
1216    pub(crate) hd_bg_addr_latch: u16,
1217    /// CHR base address of the BG tile currently feeding the shifters' high
1218    /// byte (the tile being displayed). See [`Self::hd_bg_addr_latch`].
1219    #[cfg(feature = "hd-pack")]
1220    pub(crate) hd_bg_addr_cur: u16,
1221    /// CHR base address of the next BG tile (shifters' low byte). Promoted to
1222    /// `hd_bg_addr_cur` on the prefetch byte-shift / per-tile boundary.
1223    #[cfg(feature = "hd-pack")]
1224    pub(crate) hd_bg_addr_next: u16,
1225    /// CHR base address fetched per sprite slot at `fetch_sprite_tile` time,
1226    /// consumed by `emit_pixel` for HD-pack sprite substitution.
1227    #[cfg(feature = "hd-pack")]
1228    pub(crate) hd_spr_addr: [u16; 8],
1229    /// The sprite's ORIGIN screen X per slot (the un-decremented `spr_x`), so
1230    /// `emit_pixel` can derive the column within the sprite for HD positioning.
1231    #[cfg(feature = "hd-pack")]
1232    pub(crate) hd_spr_x: [u8; 8],
1233    /// The sprite's flip-baked texel ROW (0..=7) per slot, captured at fetch (the
1234    /// `emit_pixel`-time scanline isn't enough to recover it post-shift).
1235    #[cfg(feature = "hd-pack")]
1236    pub(crate) hd_spr_off_y: [u8; 8],
1237    /// Absolute CHR-ROM tile index (`chr_phys/16`, or [`HD_CHR_RAM`]) tracked in
1238    /// lock-step with the `hd_bg_addr_*` cascade — the CHR-ROM HD-pack key.
1239    #[cfg(feature = "hd-pack")]
1240    pub(crate) hd_bg_idx_latch: u32,
1241    /// CHR-ROM tile index of the BG tile feeding the shifters' high byte.
1242    #[cfg(feature = "hd-pack")]
1243    pub(crate) hd_bg_idx_cur: u32,
1244    /// CHR-ROM tile index of the next BG tile (shifters' low byte).
1245    #[cfg(feature = "hd-pack")]
1246    pub(crate) hd_bg_idx_next: u32,
1247    /// Absolute CHR-ROM tile index per sprite slot (or [`HD_CHR_RAM`]).
1248    #[cfg(feature = "hd-pack")]
1249    pub(crate) hd_spr_idx: [u32; 8],
1250
1251    /// v2.3.2 "Lucid" — per-byte write attribution for CIRAM / OAM / palette RAM.
1252    ///
1253    /// `None` (the default) until the frontend arms it via
1254    /// [`Self::set_write_attribution`], so an unarmed `debug-hooks` build pays one
1255    /// `Option` test per PPU-memory write and no heap at all. Output-only
1256    /// telemetry: nothing in the emulation path reads it, so an armed store is
1257    /// bit-identical to an unarmed one. Deliberately NOT part of the save-state —
1258    /// attribution describes writes *this session* performed, and a restored
1259    /// state's bytes have no such history (see [`crate::provenance`]).
1260    #[cfg(feature = "debug-hooks")]
1261    pub(crate) write_attrib: Option<Box<crate::provenance::WriteAttribution>>,
1262    /// The `(pc, cycle)` of the CPU instruction currently executing, pushed down
1263    /// by the bus before each `$2000-$3FFF` register write and before each OAM
1264    /// DMA burst. Stamped into every attribution record made while it is set.
1265    ///
1266    /// The PPU cannot derive this itself: it never sees the CPU's program
1267    /// counter, and the effective VRAM destination the bus would need to record
1268    /// the attribution itself lives in the PPU's internal `v` register. Splitting
1269    /// the two halves this way is what lets each side contribute only what it
1270    /// actually knows.
1271    #[cfg(feature = "debug-hooks")]
1272    pub(crate) attrib_pc: u16,
1273    /// CPU cycle counterpart of [`Self::attrib_pc`].
1274    #[cfg(feature = "debug-hooks")]
1275    pub(crate) attrib_cycle: u64,
1276    /// The `(pc, cycle)` of the `STA $4014` that armed the in-flight OAM DMA.
1277    ///
1278    /// Separate from [`Self::attrib_pc`] because the burst does not run during
1279    /// the triggering instruction: `$4014` only sets `dma_pending`, and the 513
1280    /// or 514 DMA cycles are then stolen from the instructions that follow. By
1281    /// the time the first byte lands, [`Self::attrib_pc`] has already advanced to
1282    /// whichever instruction is being halted — an answer that is true about the
1283    /// *timing* and wrong about the *cause*. The bus latches this pair at the
1284    /// `$4014` write via [`Self::latch_dma_attrib_context`] so every byte of the
1285    /// burst names the store that actually caused it.
1286    #[cfg(feature = "debug-hooks")]
1287    pub(crate) dma_attrib_pc: u16,
1288    /// CPU cycle counterpart of [`Self::dma_attrib_pc`].
1289    #[cfg(feature = "debug-hooks")]
1290    pub(crate) dma_attrib_cycle: u64,
1291
1292    /// v2.3.2 "Lucid" phase 2 — per-pixel provenance for the current frame.
1293    ///
1294    /// `None` until armed, like [`Self::write_attrib`]. Overwritten in place
1295    /// every frame, exactly like the framebuffer it shadows.
1296    #[cfg(feature = "debug-hooks")]
1297    pub(crate) prov_frame: Option<Box<crate::provenance::PixelProvenanceFrame>>,
1298    /// Fast "is provenance armed?" flag, mirroring `prov_frame.is_some()`.
1299    ///
1300    /// `emit_pixel` runs 61,440 times a frame and is one of the two hottest
1301    /// functions in the emulator, so the per-pixel guard is a plain `bool` load
1302    /// rather than an `Option<Box<..>>` discriminant behind a pointer. Same
1303    /// shape as the bus's `event_logging` / `access_logging` flags.
1304    #[cfg(feature = "debug-hooks")]
1305    pub(crate) prov_armed: bool,
1306    /// Nametable address of the most recent NT fetch, awaiting commit.
1307    ///
1308    /// Held separately from [`Self::prov_bg_latch`] because the PPU performs two
1309    /// **dummy nametable fetches** at dots 337-340, after the pre-render line's
1310    /// last real tile has been fetched but before the visible line's first
1311    /// reload consumes it. Writing the NT address straight into the latch let
1312    /// those dummies overwrite the pending tile, so the first visible tile group
1313    /// reported the address of the tile after it. A tile is defined when its
1314    /// PATTERN is fetched — which the dummy fetches never do — so the pending
1315    /// addresses are committed in `fetch_bg_lo`.
1316    #[cfg(feature = "debug-hooks")]
1317    pub(crate) prov_nt_pending: u16,
1318    /// Attribute address awaiting commit. See [`Self::prov_nt_pending`].
1319    #[cfg(feature = "debug-hooks")]
1320    pub(crate) prov_at_pending: u16,
1321    /// Addresses of the background tile most recently FETCHED (the tile two
1322    /// slots ahead of the one on screen). Committed at pattern-fetch time;
1323    /// promoted through `next` into `cur` by the same shift-register reloads
1324    /// that move the pattern bytes, so the cascade stays tile-for-tile aligned
1325    /// with what the shifters are emitting.
1326    #[cfg(feature = "debug-hooks")]
1327    pub(crate) prov_bg_latch: ProvBgAddrs,
1328    /// Addresses of the background tile currently BEING DISPLAYED — the one
1329    /// `emit_pixel` must report.
1330    ///
1331    /// This cascade exists because `v` cannot answer the question: by the time a
1332    /// tile's pixels reach the screen, `v` has already advanced two tiles past
1333    /// it. Deriving the nametable address from `v` at emit time would be wrong
1334    /// for every pixel, and wrong in a way that looks plausible.
1335    #[cfg(feature = "debug-hooks")]
1336    pub(crate) prov_bg_cur: ProvBgAddrs,
1337    /// Addresses of the next background tile (the shifters' low byte), promoted
1338    /// into [`Self::prov_bg_cur`] on the per-tile boundary.
1339    #[cfg(feature = "debug-hooks")]
1340    pub(crate) prov_bg_next: ProvBgAddrs,
1341    /// Pattern address fetched per sprite slot, so a sprite pixel can report the
1342    /// CHR row behind it. Mirrors the `hd-pack` `hd_spr_addr` capture, kept
1343    /// separate so neither feature's telemetry depends on the other being on.
1344    #[cfg(feature = "debug-hooks")]
1345    pub(crate) prov_spr_addr: [u16; 8],
1346}
1347
1348/// The three VRAM addresses that produced one background tile.
1349///
1350/// Carried as a unit through the PPU's internal fetch → display cascade
1351/// (`latch` → `next` → `cur`), so a promotion is one struct copy instead of
1352/// three separate field moves that could drift out of step with each other.
1353#[cfg(feature = "debug-hooks")]
1354#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
1355pub struct ProvBgAddrs {
1356    /// Nametable address the tile number came from.
1357    pub nt: u16,
1358    /// Attribute address the palette group came from. Carried rather than
1359    /// derived, because an MMC5 vertical split supplies its own attribute
1360    /// address that the standard `$23C0 | ...` arithmetic cannot produce.
1361    pub at: u16,
1362    /// CHR address of the pattern row (the low-plane address; the high plane is
1363    /// `+8`).
1364    pub pattern: u16,
1365}
1366
1367/// v2.0 Phase 6 (`mc-ppu-subpos`): the analog `$2001` PPUMASK write delay.
1368///
1369/// In PPU dots. `TriCNES` applies a `$2001` write 2-3 dots after the CPU write
1370/// (`PPU_Update2001Delay`, sub-dot alignment dependent); this emulator's reload
1371/// sits one dot later than that plus the existing 1-dot render gate, so the
1372/// effective default is 4. Runtime-tunable (an atomic) so the exact phase can be
1373/// swept against the `BG Serial In` / `Stale BG Shift` keys without rebuilding.
1374/// The `mc-ppu-subpos` name in the lines above is the v2.0 Phase 6 WORKSTREAM,
1375/// not a cargo feature: no manifest declares one, and the read at the BG-reload
1376/// freeze below is unconditional. Kept as the historical label rather than
1377/// rewritten, but do not go looking for a flag to turn on.
1378pub static MASK_WRITE_DELAY: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(4);
1379
1380/// v2.6.18 DERIVATION KNOB: depth, in PPU dots, of the rendering-enable
1381/// pipeline (`rendering_enabled_delayed`).
1382///
1383/// 1 = shipped semantics exactly (a consumer sees the PREVIOUS dot's rendering
1384/// value). 0 = no delay, consumers see the live mask. 2 = two dots.
1385///
1386/// This delay and the write-commit point are ONE quantity split across two
1387/// places: the observable effect time is the commit offset PLUS this depth.
1388/// v2.6.17 swept this 1..4 and found 1 optimal -- but only at the SHIPPED commit
1389/// point. Under phi2 the commit is half a dot later, so the depth reproducing
1390/// the same effect time is one LESS, exactly as `mask_for_skip_check` needed
1391/// 2 -> 1. Sweeping at the shipped placement answers a different question.
1392///
1393/// Feature-gated; the shipped build has no atomic load on the per-dot path.
1394#[cfg(feature = "phi2-write-sweep")]
1395pub static RENDER_GATE_LAG: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(1);
1396
1397/// Whether the specialized dot paths may be taken this dot.
1398///
1399/// Their guards prove a ONE-dot rendering history (`rendering_enabled_delayed`
1400/// and `prev_rendering_enabled`), which is exactly what depths 0 and 1 read. At
1401/// depth >= 2 the gate reads [`Ppu::render_gate_prev2`] — rendering from two
1402/// dots ago — and the guard says nothing about it, so a dot with rendering
1403/// enabled for the last two dots but disabled before them would run the
1404/// rendering-enabled fast body while the general path gates it off. Excluding
1405/// the fast paths at depth >= 2 keeps one code path answering the sweep's
1406/// question instead of two that disagree.
1407// Dead under `ppu-state-trace` for the same reason `tick_visible_render_fast`
1408// is: that feature `cfg`s the fast-path dispatch out entirely, so the guard
1409// term this function supplies has no caller. Live in every other build --
1410// the one shape that earns an `allow` rather than a deletion, and scoped to
1411// that feature so it cannot suppress a real finding elsewhere.
1412#[cfg(feature = "phi2-write-sweep")]
1413#[cfg_attr(feature = "ppu-state-trace", allow(dead_code))]
1414#[inline]
1415fn fast_dot_paths_valid() -> bool {
1416    // The OAM2 pipeline is shifted only on the general path, and the fast
1417    // bodies read the gate, so an armed OAM2 knob must take the general path
1418    // for the same reason a depth-2 render gate must. Learned the expensive
1419    // way on #506: a bypassed pipeline does not merely go stale, it makes the
1420    // swept cell measure something that is not the configuration named.
1421    // `SCROLL_GATE_LAG` joins them for a THIRD instance of the same trap: the
1422    // visible fast body performs its own dot-256 `inc_vert_v()`, so an armed
1423    // scroll knob is bypassed entirely and every swept cell reads identical --
1424    // which is exactly what the first run of that sweep reported, four depths
1425    // all at 143/144 with nothing gained and nothing lost.
1426    RENDER_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) <= 1
1427        && OAM2_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) == 2
1428        && SCROLL_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) == u8::MAX
1429}
1430
1431/// Default build: the depth is the shipped 1, so the guards are sufficient and
1432/// this folds away at compile time — the fast paths are reached exactly as
1433/// before.
1434#[cfg(not(feature = "phi2-write-sweep"))]
1435#[inline]
1436const fn fast_dot_paths_valid() -> bool {
1437    true
1438}
1439
1440/// The default build's guard term must be a compile-time `true`, or adding it
1441/// would have changed which dots take the specialized paths — and those paths
1442/// exist to be byte-identical. Asserted at compile time rather than in a test,
1443/// because "it folds away" is a property of the constant, not of a run.
1444#[cfg(not(feature = "phi2-write-sweep"))]
1445const _: () = assert!(fast_dot_paths_valid());
1446/// v2.6.18 condition-2 DIAGNOSTIC: the dot on scanline 241 that sets the VBL
1447/// flag. Default 1, which is what nesdev specifies and what ships.
1448///
1449/// Exists to separate two explanations for the six NMI entries failing when the
1450/// CPU access moves to the cycle's last dot: a pure one-dot relative shift
1451/// between the `$2002` read and VBL-set, versus a defect in the read placement
1452/// itself. Moving this is NOT a fix and must never be shipped non-default.
1453#[cfg(feature = "phi2-write-sweep")]
1454pub static VBL_SET_DOT: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(1);
1455
1456/// v2.6.18 condition-2 DIAGNOSTIC: the depth of the odd-frame-skip gate's
1457/// `$2001` pipeline, in stages. Default 2, which is what ships.
1458///
1459/// The skip gate is the most explicitly-documented compensation in this PPU --
1460/// its own comment says lockstep applies the write at the START of a CPU cycle
1461/// while hardware latches at phi2 -- so its depth is coupled to where the write
1462/// access lands, and sweeping the two independently answers a different
1463/// question than sweeping them together.
1464#[cfg(feature = "phi2-write-sweep")]
1465pub static SKIP_GATE_LAG: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(2);
1466
1467/// v2.6.18: the depth of the dot-256 vertical increment's own `$2001` gate.
1468///
1469/// `u8::MAX` (the default) means "ride the shared `rendering_gate`", which is
1470/// what ships. This is the consumer `Frozen OAM2 Increment` turns on: the ROM
1471/// enables rendering ON dot 256 and states that the vertical scroll is NOT
1472/// incremented, and our enable lands a dot early.
1473#[cfg(feature = "phi2-write-sweep")]
1474pub static SCROLL_GATE_LAG: core::sync::atomic::AtomicU8 =
1475    core::sync::atomic::AtomicU8::new(u8::MAX);
1476
1477/// v2.6.18: the depth of the dot-339 sprite-counter re-arm's own `$2001` gate.
1478///
1479/// `u8::MAX` (the default) means "ride the shared `rendering_gate`", which is
1480/// what ships. Any other value gives the re-arm its own depth in the shared
1481/// history, which is the question `Stale Sprite Shift Regs` and
1482/// `Frozen OAM2 Increment` forced: at the shipped placement the frozen-flag
1483/// machinery closes only at `RENDER_GATE_LAG = 2`, and that same depth is what
1484/// breaks the re-arm, which wants 1. Two consumers, one gate, opposite
1485/// requirements -- the shape v2.6.5 resolved by separating the background
1486/// shifters' reload from their shift clock.
1487#[cfg(feature = "phi2-write-sweep")]
1488pub static SPRITE_REARM_LAG: core::sync::atomic::AtomicU8 =
1489    core::sync::atomic::AtomicU8::new(u8::MAX);
1490
1491/// v2.6.18: the depth of the OAM2 counter's `$2001` gate, in dots. Default 0 =
1492/// the live mask, which is what ships.
1493///
1494/// Every other rendering consumer here is delayed and this one is not, so the
1495/// two OAM2 catalog entries -- which share it -- can only be satisfied at
1496/// different CPU access placements. Sweeping the depth asks whether that is the
1497/// reason, which no placement sweep can. Depth 3 is what closes
1498/// `Frozen OAM2 Increment`; the borrowed two-stage pipeline it first used could
1499/// not reach that, which is why the field below is dedicated.
1500#[cfg(feature = "phi2-write-sweep")]
1501pub static OAM2_GATE_LAG: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(2);
1502
1503impl Ppu {
1504    /// New PPU in power-on state.
1505    #[must_use]
1506    // The power-on field initialization is naturally long (the PPU has many
1507    // state fields); the feature-gated `hd-pack` initializers nudge it over the
1508    // 100-line lint. Splitting the struct literal would hurt readability.
1509    #[allow(clippy::too_many_lines)]
1510    pub fn new(region: PpuRegion) -> Self {
1511        let mut p = Self {
1512            region,
1513            ctrl: PpuCtrl::empty(),
1514            mask: PpuMask::empty(),
1515            mask_for_skip_check: PpuMask::empty(),
1516            dot0_replaced: false,
1517            mask_skip_pipe1: PpuMask::empty(),
1518            status: PpuStatus::empty(),
1519            oam_addr: 0,
1520            data_buffer: 0,
1521            render_data_bus: 0,
1522            ppudata_sm_countdown: 0,
1523            ppudata_v_inc_pending: false,
1524            spr_fetch_lo_raw: [0; 8],
1525            spr_fetch_hi_raw: [0; 8],
1526            ppudata_spr0_nt_addr: 0x2000,
1527            octal_latch: 0,
1528            address_bus: 0,
1529            ale_armed: false,
1530            pattern_latch_stale: false,
1531            copy_v_delay: 0,
1532            v: 0,
1533            t: 0,
1534            x: 0,
1535            w: false,
1536            ciram: vec![0u8; 0x0800].into_boxed_slice(),
1537            oam: vec![0u8; 0x0100].into_boxed_slice(),
1538            secondary_oam: [0xFF; 32],
1539            oam_bus_copybuffer: 0xFF,
1540            oam_bus_secondary: [0xFF; 32],
1541            oam_bus_addr_h: 0,
1542            oam_bus_addr_l: 0,
1543            oam_bus_secondary_addr: 0,
1544            oam_bus_copy_done: false,
1545            oam_bus_sprite_in_range: false,
1546            oam_bus_overflow_counter: 0,
1547            palette_ram: [0u8; 32],
1548            // v2.1.4 F2.3 — OAM decay is off by default; the timestamps start at 0
1549            // (row "last touched at cycle 0"). While disabled they are never read,
1550            // and `set_oam_decay(true)` re-bases them to the current cycle so
1551            // enabling mid-run does not instantly decay every row.
1552            oam_decay_cycles: [0; 32],
1553            oam_decay_enabled: false,
1554            open_bus: 0,
1555            open_bus_decay: [0; 3],
1556            nmi_line: false,
1557            suppress_vbl_this_frame: false,
1558            last_a12_level: false,
1559            // Power-up position matches Mesen2's NesPpu::Reset(false) endpoint
1560            // (_scanline=-1, _cycle=340). After the first PPU tick, wraps to
1561            // (scanline=0, dot=0, frame+=1), putting the post-power-on PPU
1562            // position within ~2 dots of Mesen2's. Combined with the 8-cycle
1563            // CPU reset (see Cpu::reset), this closes the +344-dot PPU offset
1564            // identified empirically in Session-13 (docs/audit/
1565            // session-13-cpu-boot-fix-2026-05-21.md).
1566            //
1567            // SESSION-29 CRITICAL FINDING: Option (a) "PPU re-baseline"
1568            // empirically attempted and DOES NOT CLOSE THE C1 AXIS.
1569            // Shifting PPU init by +2 dots to (scanline=0, dot=1):
1570            //   - Generates 24 snapshot regressions (audio_db, visual,
1571            //     m22, Cascade A) — all are "expected" cosmetic shifts.
1572            //   - BUT the cpu_interrupts_v2/{2,3,5}_strict probes STILL
1573            //     FAIL — confirmed via `cargo test ... --include-ignored`.
1574            //
1575            // The +2 dot shift moves everything uniformly: VBL set position
1576            // AND BIT $2002 read position both shift by +2 dots, preserving
1577            // the relative race-window relationship.  The BIT $2002 polling
1578            // loop inside blargg `sync_vbl` still hits the pre-VBL-set side
1579            // of the race window.
1580            //
1581            // CONCLUSION: closing C1 requires changing the PHASE
1582            // RELATIONSHIP between CPU and PPU (Option b — master-clock-
1583            // precise scheduling refactor), NOT a global PPU init shift.
1584            // The 4 C1 IRQ-timing residuals are deferred to v2.0 with the
1585            // master-clock refactor; v1.0.0 ships at 90.65% AccuracyCoin
1586            // with the 4 residuals documented as v2.0-deferred.  See
1587            // `docs/audit/session-29-c1-axis-final-conclusion-2026-05-23.md`
1588            // + `docs/audit/session-29-option-a-empirical-falsification.md`.
1589            dot: 340,
1590            scanline: region.prerender_line(),
1591            frame: 0,
1592            frame_complete: false,
1593            post_reset_mask_remaining: region.post_reset_mask_cycles(),
1594            nt_latch: 0,
1595            at_latch: 0,
1596            bg_lo_latch: 0,
1597            bg_hi_latch: 0,
1598            bg_shift_lo: 0,
1599            bg_shift_hi: 0,
1600            at_shift_lo: 0,
1601            at_shift_hi: 0,
1602            ex_attr_latch: None,
1603            bg_split_latch: None,
1604            spr_shift_lo: [0; 8],
1605            spr_shift_hi: [0; 8],
1606            spr_attr: [0; 8],
1607            spr_x: [0; 8],
1608            spr_halted: [true; 8],
1609            spr_count: 0,
1610            spr_zero_in_line: false,
1611            sprite_eval_read_latch: 0xFF,
1612            sprite_eval_n: 0,
1613            sprite_eval_m: 0,
1614            sprite_eval_found: 0,
1615            sprite_eval_sec_idx: 0,
1616            sprite_eval_copying: false,
1617            sprite_eval_done: false,
1618            sprite_eval_overflow_search: false,
1619            sprite_eval_zero_found: false,
1620            sprite_eval_first_iter: false,
1621            oam_corruption_pending: false,
1622            oam_corruption_index: 0,
1623            oam_corruption_disabled: false,
1624            oam_corruption_disabled_instant: false,
1625            oam2_addr: 0,
1626            oam2_fetch_addr: 0,
1627            oam2_overflowed: false,
1628            oam2_fetch_frozen: false,
1629            prev_rendering_enabled: false,
1630            rendering_enabled_delayed: false,
1631            #[cfg(feature = "phi2-write-sweep")]
1632            render_gate_prev2: false,
1633            rendering_enabled_delayed2: false,
1634            spr_rearm_deferred: false,
1635            #[cfg(feature = "phi2-write-sweep")]
1636            sweep_mask_history: [PpuMask::empty(); 4],
1637            bg_reload_render: false,
1638            mask_write_delay: 0,
1639            cached_visible: false,
1640            cached_pre_render: false,
1641            cached_render_line: false,
1642            #[cfg(feature = "ppu-idle-line-fast")]
1643            cached_idle_line: false,
1644            flags_cached_scanline: i16::MIN,
1645            active_palette: crate::palette::PpuPalette::Composite2C02,
1646            rgba_lut: build_rgba_lut(crate::palette::PpuPalette::Composite2C02),
1647            custom_palette: None,
1648            is_2c05: false,
1649            id_2c05: 0,
1650            // v2.1.7 P5 — default revision models no extra corruption; default
1651            // power-up palette is all-zero. Both keep the default build
1652            // byte-identical (the `palette_ram: [0u8; 32]` above already reflects
1653            // the `PaletteInit::Zeroed` default).
1654            die_revision: PpuRevision::Rp2c02H,
1655            power_up_palette: PaletteInit::Zeroed,
1656            framebuffer: vec![0u8; FRAMEBUFFER_LEN].into_boxed_slice(),
1657            index_framebuffer: vec![0u16; FRAMEBUFFER_PIXELS].into_boxed_slice(),
1658            #[cfg(feature = "ppu-fetch-trace")]
1659            fast_path_hits: 0,
1660            dot_counter: 0,
1661            frame_ntsc_phase: 0,
1662            extra_scanlines: 0,
1663            sprite_limit_disabled: false,
1664            spr_extra_count: 0,
1665            spr_extra_lo: [0; MAX_EXTRA_SPRITES],
1666            spr_extra_hi: [0; MAX_EXTRA_SPRITES],
1667            spr_extra_attr: [0; MAX_EXTRA_SPRITES],
1668            spr_extra_x: [0; MAX_EXTRA_SPRITES],
1669            extra_lines_remaining: 0,
1670            // v2.2.3 performance pass: promoted to the default (was `false`
1671            // through v2.2.2). Byte-identical to the exact path by
1672            // construction and by `fast_dotloop_diff.rs`; -11.3% on the
1673            // rendering-heavy `full_frame` bench. See the field's rustdoc.
1674            fast_dotloop: true,
1675            #[cfg(feature = "ppu-state-trace")]
1676            state_trace: None,
1677            #[cfg(feature = "ppu-state-trace")]
1678            trace_cpu_cycle: 0,
1679            #[cfg(feature = "ppu-fetch-trace")]
1680            fetch_trace: None,
1681            #[cfg(feature = "hd-pack")]
1682            hd_tile_source: vec![HdTileSource::default(); FRAMEBUFFER_PIXELS].into_boxed_slice(),
1683            #[cfg(feature = "hd-pack")]
1684            hd_bg_addr_latch: HD_TILE_NONE,
1685            #[cfg(feature = "hd-pack")]
1686            hd_bg_addr_cur: HD_TILE_NONE,
1687            #[cfg(feature = "hd-pack")]
1688            hd_bg_addr_next: HD_TILE_NONE,
1689            #[cfg(feature = "hd-pack")]
1690            hd_spr_addr: [HD_TILE_NONE; 8],
1691            #[cfg(feature = "hd-pack")]
1692            hd_spr_x: [0; 8],
1693            #[cfg(feature = "hd-pack")]
1694            hd_spr_off_y: [0; 8],
1695            #[cfg(feature = "hd-pack")]
1696            hd_bg_idx_latch: HD_CHR_RAM,
1697            #[cfg(feature = "hd-pack")]
1698            hd_bg_idx_cur: HD_CHR_RAM,
1699            #[cfg(feature = "hd-pack")]
1700            hd_bg_idx_next: HD_CHR_RAM,
1701            #[cfg(feature = "hd-pack")]
1702            hd_spr_idx: [HD_CHR_RAM; 8],
1703            #[cfg(feature = "debug-hooks")]
1704            write_attrib: None,
1705            #[cfg(feature = "debug-hooks")]
1706            attrib_pc: 0,
1707            #[cfg(feature = "debug-hooks")]
1708            attrib_cycle: 0,
1709            #[cfg(feature = "debug-hooks")]
1710            dma_attrib_pc: 0,
1711            #[cfg(feature = "debug-hooks")]
1712            dma_attrib_cycle: 0,
1713            #[cfg(feature = "debug-hooks")]
1714            prov_frame: None,
1715            #[cfg(feature = "debug-hooks")]
1716            prov_armed: false,
1717            #[cfg(feature = "debug-hooks")]
1718            prov_nt_pending: 0,
1719            #[cfg(feature = "debug-hooks")]
1720            prov_at_pending: 0,
1721            #[cfg(feature = "debug-hooks")]
1722            prov_bg_latch: ProvBgAddrs::default(),
1723            #[cfg(feature = "debug-hooks")]
1724            prov_bg_cur: ProvBgAddrs::default(),
1725            #[cfg(feature = "debug-hooks")]
1726            prov_bg_next: ProvBgAddrs::default(),
1727            // The "no pattern" sentinel, not 0 — 0 is a legitimate CHR address.
1728            // Unreachable at emit time today (a sprite is only selected when
1729            // `spr_idx != 0`, which implies a real fetch), but the adjacent
1730            // `hd_spr_addr` uses its own sentinel for exactly this reason and a
1731            // future reader should not have to re-derive why 0 was safe.
1732            #[cfg(feature = "debug-hooks")]
1733            prov_spr_addr: [crate::provenance::PATTERN_ADDR_NONE; 8],
1734        };
1735        // Clear status flags that match power-on per nesdev wiki: VBL is
1736        // unspecified on power-on. We start clear.
1737        p.status = PpuStatus::empty();
1738        p
1739    }
1740
1741    /// Configure the PPU's hardware variant for Vs. System / PlayChoice-10
1742    /// arcade carts.
1743    ///
1744    /// `palette` selects the output palette (the RGB PPUs replace the 2C02
1745    /// composite palette with a fixed hardware RGB lookup). `is_2c05` enables
1746    /// the 2C05's register quirks: a write to `$2000` sets MASK and a write to
1747    /// `$2001` sets CTRL (swapped), and a `$2002` read ORs `id` into its low
1748    /// bits. For a 2C02 (the default NES/Famicom path) this is never called, so
1749    /// `active_palette` stays [`crate::palette::PpuPalette::Composite2C02`],
1750    /// `is_2c05` stays `false`, and behaviour is byte-for-byte unchanged.
1751    pub const fn set_palette(
1752        &mut self,
1753        palette: crate::palette::PpuPalette,
1754        is_2c05: bool,
1755        id: u8,
1756    ) {
1757        self.active_palette = palette;
1758        // v2.8.0 Phase 4 — keep the per-pixel RGBA lookup in sync with the active
1759        // palette (byte-identical by construction; see `rgba_lut`). A loaded `.pal`
1760        // (`custom_palette`) overrides the built-in table; `rebuild_rgba_lut`
1761        // honours it.
1762        self.rebuild_rgba_lut();
1763        self.is_2c05 = is_2c05;
1764        self.id_2c05 = id;
1765    }
1766
1767    /// v1.1.0 beta.1 (T-110-A3) — install (or clear with `None`) a custom 64-entry
1768    /// base palette from a loaded `.pal` file and rebuild the RGBA lookup. `None`
1769    /// restores the built-in palette for the active [`crate::palette::PpuPalette`]
1770    /// (byte-identical to default). A frontend presentation override.
1771    pub const fn set_custom_palette(&mut self, base: Option<[[u8; 3]; 64]>) {
1772        self.custom_palette = base;
1773        self.rebuild_rgba_lut();
1774    }
1775
1776    /// v3.1.0 (`T-SPRITE-LIMIT`) — draw the sprites beyond the eighth on a
1777    /// scanline (`true`), or not (`false`, the default and the hardware). See
1778    /// the field for what stays exact. Turning it off drops any extras already
1779    /// fetched, so the next scanline draws exactly eight.
1780    pub const fn set_sprite_limit_disabled(&mut self, disabled: bool) {
1781        self.sprite_limit_disabled = disabled;
1782        if !disabled {
1783            self.spr_extra_count = 0;
1784        }
1785    }
1786
1787    /// v3.1.0 — whether the sprites beyond the eighth are drawn.
1788    #[must_use]
1789    pub const fn sprite_limit_disabled(&self) -> bool {
1790        self.sprite_limit_disabled
1791    }
1792
1793    /// v3.1.0 — how many sprites beyond the eighth are fetched for the next
1794    /// scanline (`0` unless [`Self::sprite_limit_disabled`]).
1795    #[must_use]
1796    pub const fn extra_sprite_count(&self) -> u8 {
1797        self.spr_extra_count
1798    }
1799
1800    /// v1.7.0 "Forge" Workstream F3 — set the number of EXTRA blank vblank
1801    /// scanlines to insert per frame (the PPU extra-scanlines overclock).
1802    ///
1803    /// `0` (the default) is stock NES timing and is **byte-identical** to a PPU
1804    /// that never calls this. A non-zero value lengthens vblank by that many
1805    /// idle scanlines each frame (more CPU run-time, no visible change), at the
1806    /// existing dot resolution. Off by default; a frontend config knob, not part
1807    /// of the save-state. Distinct from the CPU-multiplier overclock (v3.1.0).
1808    ///
1809    /// Changing the count cancels any in-flight insertion for the current
1810    /// frame: the per-frame countdown (`extra_lines_remaining`) is
1811    /// reset to `0` so it cannot remain stale or out-of-bounds relative to
1812    /// the new `lines` (e.g. shrinking 8 → 2, or disabling N → 0). The next
1813    /// frame reloads the countdown from the new value at the insertion point.
1814    pub const fn set_extra_scanlines(&mut self, lines: u16) {
1815        self.extra_scanlines = lines;
1816        self.extra_lines_remaining = 0;
1817    }
1818
1819    /// v1.7.0 F3 — the currently-configured extra-scanline count (`0` = stock).
1820    #[must_use]
1821    pub const fn extra_scanlines(&self) -> u16 {
1822        self.extra_scanlines
1823    }
1824
1825    /// v2.1.8 A1 — enable/disable the specialized visible-scanline fast dot
1826    /// path. **Default ON since the v2.2.3 performance pass** (was OFF through
1827    /// v2.2.2); either setting produces the identical frame, so this selects a
1828    /// code path, not a behaviour. See [`Self::fast_dotloop`] and
1829    /// `docs/performance.md`.
1830    pub const fn set_fast_dotloop(&mut self, enabled: bool) {
1831        self.fast_dotloop = enabled;
1832    }
1833
1834    /// v2.1.8 A1 — whether the visible-scanline fast dot path is enabled.
1835    #[must_use]
1836    pub const fn fast_dotloop(&self) -> bool {
1837        self.fast_dotloop
1838    }
1839
1840    /// v1.1.0 beta.1 — the custom 64-entry base palette installed by
1841    /// [`Self::set_custom_palette`], or `None` for the built-in one.
1842    #[must_use]
1843    pub const fn custom_palette(&self) -> Option<[[u8; 3]; 64]> {
1844        self.custom_palette
1845    }
1846
1847    /// v2.9.8 — carry the host's settings from `prev` onto this freshly built
1848    /// PPU, so a power cycle (which rebuilds the PPU from [`Self::new`]) keeps
1849    /// them.
1850    ///
1851    /// # Why this lives on the PPU
1852    ///
1853    /// A power cycle is a cold boot of the console, not of the user's
1854    /// configuration. Until v2.9.8 the bus rebuilt the PPU and re-applied only
1855    /// the settings it also stored itself (the die revision, the power-up
1856    /// palette, the Vs. RGB palette), so every setting held ONLY here --
1857    /// the custom / generated palette, the extra-scanlines overclock, the fast
1858    /// dot path selector and the OAM-decay model -- silently reverted to its
1859    /// default, and each host had to remember to push it again. The list of
1860    /// what a setting is belongs next to the fields, so a new one is carried
1861    /// where it is declared; `snapshot_schema_audit.rs` cross-checks it against
1862    /// the fields that audit classifies as configuration.
1863    ///
1864    /// # What is carried, and what is not
1865    ///
1866    /// Carried: [`Self::custom_palette`] (the lookup table is rebuilt to
1867    /// honour it), [`Self::extra_scanlines`], [`Self::sprite_limit_disabled`],
1868    /// [`Self::fast_dotloop`] and
1869    /// [`Self::oam_decay_enabled`]. The decay switch goes through
1870    /// [`Self::set_oam_decay`], exactly as a host enabling it on a fresh
1871    /// console would, so the result is what a fresh boot with the setting
1872    /// applied produces.
1873    ///
1874    /// Not carried here: the die revision and the power-up palette are stored
1875    /// on the bus, which re-applies them (the power-up palette is a power-on
1876    /// FILL and must be rewritten, not copied); the active palette and the
1877    /// 2C05 identity are board identity, re-derived from the cartridge by the
1878    /// bus. The state and fetch traces are capture buffers, not settings: a
1879    /// cold boot ends the history they describe, as it does for the
1880    /// provenance stores, which the core moves across separately (armed, then
1881    /// emptied).
1882    ///
1883    /// Every carried value is a selector or an output override, so with
1884    /// every setting at its default this leaves the PPU byte-identical to
1885    /// [`Self::new`].
1886    pub const fn adopt_settings_from(&mut self, prev: &Self) {
1887        self.custom_palette = prev.custom_palette;
1888        self.rebuild_rgba_lut();
1889        self.set_extra_scanlines(prev.extra_scanlines);
1890        self.sprite_limit_disabled = prev.sprite_limit_disabled;
1891        self.fast_dotloop = prev.fast_dotloop;
1892        self.set_oam_decay(prev.oam_decay_enabled);
1893    }
1894
1895    /// v2.1.4 F2.3 — enable or disable the optional OAM-decay accuracy model.
1896    ///
1897    /// **Off by default.** When off (the default) OAM reads/writes never consult
1898    /// the decay state and the deterministic output is **byte-identical** to a
1899    /// build without the feature — `AccuracyCoin`, the commercial oracle, and the
1900    /// visual/`external_real_games` regression suites are unaffected. When on, the
1901    /// PPU refreshes each 8-byte OAM row on every read (sprite evaluation + `$2004`)
1902    /// and write; a row that goes un-refreshed for more than `OAM_DECAY_CPU_CYCLES`
1903    /// (3000) CPU cycles decays to Mesen2's canonical garbage pattern on the next
1904    /// read (`oam_decay_on_read`). The model is NTSC/Dendy-only (PAL's refresh
1905    /// cadence masks decay) — the region gate lives in the hooks, so it is safe to
1906    /// enable on any region.
1907    ///
1908    /// A frontend/config knob (re-applied on load like `region` / `active_palette`),
1909    /// **not** part of the save-state. Turning the model ON re-bases every row's
1910    /// timestamp to the current CPU cycle so a freshly-enabled model does not report
1911    /// every row as instantly decayed; turning it OFF leaves the timestamps as-is
1912    /// (they are simply no longer consulted).
1913    pub const fn set_oam_decay(&mut self, enabled: bool) {
1914        if enabled && !self.oam_decay_enabled {
1915            // Freshly enabling: treat every row as just-refreshed so the first
1916            // post-enable reads don't spuriously report a multi-second-old row as
1917            // decayed. Idempotent re-enables (already on) skip this so a long
1918            // rendering-disabled span already in progress keeps decaying.
1919            let now = self.dot_counter / 3;
1920            let mut i = 0;
1921            while i < self.oam_decay_cycles.len() {
1922                self.oam_decay_cycles[i] = now;
1923                i += 1;
1924            }
1925        }
1926        self.oam_decay_enabled = enabled;
1927    }
1928
1929    /// v2.1.4 F2.3 — whether the optional OAM-decay model is currently enabled.
1930    #[must_use]
1931    pub const fn oam_decay_enabled(&self) -> bool {
1932        self.oam_decay_enabled
1933    }
1934
1935    /// v2.1.7 P5 — select the emulated 2C02 die revision (see [`PpuRevision`]).
1936    ///
1937    /// The [`PpuRevision::default`] ([`PpuRevision::Rp2c02H`]) models no extra
1938    /// quirks, so at the default this is behaviorally inert and the PPU is
1939    /// byte-identical to a build without the field. Selecting
1940    /// [`PpuRevision::Rp2c02G`] additionally arms the OAMADDR (`$2003`)
1941    /// write-during-rendering OAM corruption glitch. A construction/config knob,
1942    /// re-applied on load like the region / active palette — not part of the
1943    /// save-state.
1944    pub const fn set_revision(&mut self, revision: PpuRevision) {
1945        self.die_revision = revision;
1946    }
1947
1948    /// v2.1.7 P5 — the currently-selected 2C02 die revision.
1949    #[must_use]
1950    pub const fn revision(&self) -> PpuRevision {
1951        self.die_revision
1952    }
1953
1954    /// v2.9.8 — CPU cycles left in the post-reset warm-up window, during which
1955    /// writes to `$2000`/`$2001`/`$2005`/`$2006` are ignored (`0` once the
1956    /// window has passed). Read-only; see [`PpuRegion::post_reset_mask_cycles`]
1957    /// and `docs/ppu-2c02.md` (§Power-up and reset).
1958    #[must_use]
1959    pub const fn warmup_cycles_remaining(&self) -> u32 {
1960        self.post_reset_mask_remaining
1961    }
1962
1963    /// v2.9.8 — end the post-reset warm-up window now, so `$2000`/`$2001`/
1964    /// `$2005`/`$2006` writes take effect from the next CPU cycle.
1965    ///
1966    /// This is how the opt-in Famicom console model is expressed at the PPU:
1967    /// the `NESdev` wiki's "PPU power up state" (§Famicom) documents that the
1968    /// Famicom ties the PPU's `/RESET` to 5 V while the CPU's `/RESET` rides a
1969    /// 0.47 µF
1970    /// capacitor, so at power-on the PPU begins initialising roughly one frame
1971    /// (about 29,781 CPU cycles) before the CPU leaves reset. The warm-up window
1972    /// is 29,658 cycles on NTSC, shorter than that frame, so by the time the
1973    /// CPU executes its first instruction the window has already closed. The
1974    /// PPU itself is unchanged; the console decides when its reset is released,
1975    /// which is why the caller (the console model in `rustynes-core`) owns the
1976    /// decision and this is only the mechanism.
1977    ///
1978    /// Only the window is touched. Every register and the frame position stay
1979    /// as they are, and the field it clears is already part of the save-state,
1980    /// so no snapshot change follows from calling it.
1981    pub const fn end_warmup(&mut self) {
1982        self.post_reset_mask_remaining = 0;
1983    }
1984
1985    /// v2.1.7 P5 — apply a power-up palette-RAM pattern (see [`PaletteInit`]).
1986    ///
1987    /// Writes all 32 palette-RAM bytes to the selected pattern and records the
1988    /// selection so a subsequent power-cycle can re-apply it. The
1989    /// [`PaletteInit::default`] ([`PaletteInit::Zeroed`]) writes all-zero — the
1990    /// established power-up state — so at the default this leaves the PPU
1991    /// byte-identical. Intended to be called at construction / power-on (palette
1992    /// RAM is not cleared on a warm reset, matching real hardware). It writes
1993    /// [`Self::palette_ram`] directly, which the snapshot already serializes, so
1994    /// no snapshot-format change is required.
1995    pub const fn apply_power_up_palette(&mut self, init: PaletteInit) {
1996        self.power_up_palette = init;
1997        match init {
1998            PaletteInit::Zeroed => {
1999                let mut i = 0;
2000                while i < self.palette_ram.len() {
2001                    self.palette_ram[i] = 0;
2002                    i += 1;
2003                }
2004            }
2005            PaletteInit::Blargg => {
2006                let mut i = 0;
2007                while i < self.palette_ram.len() {
2008                    // Palette-RAM cells are 6-bit; mask to match a `$2007` write.
2009                    self.palette_ram[i] = BLARGG_POWER_UP_PALETTE[i] & 0x3F;
2010                    i += 1;
2011                }
2012            }
2013        }
2014    }
2015
2016    /// v2.1.7 P5 — the currently-selected power-up palette pattern.
2017    #[must_use]
2018    pub const fn power_up_palette(&self) -> PaletteInit {
2019        self.power_up_palette
2020    }
2021
2022    /// v2.1.4 F2.3 — `true` when the OAM-decay model should act this access:
2023    /// enabled AND the region is NTSC/Dendy (PAL's frequent refresh masks decay,
2024    /// so Mesen2 never decays there). This is the single gate every decay hook
2025    /// funnels through; at the default (disabled) it is a single bool test and the
2026    /// hooks are behaviour-neutral.
2027    #[inline]
2028    const fn oam_decay_active(&self) -> bool {
2029        self.oam_decay_enabled && !matches!(self.region, PpuRegion::Pal)
2030    }
2031
2032    /// v2.1.4 F2.3 — OAM-read decay hook. Call **immediately before** reading
2033    /// `oam[addr]` at every primary-OAM read site (the `$2004` read and both
2034    /// sprite-evaluation read paths). Implements the `NESdev`-documented OAM DRAM
2035    /// decay-on-read behavior (`NESdev` wiki "PPU OAM" — sprite RAM is dynamic and
2036    /// its cells decay; a read recharges the touched row):
2037    ///
2038    /// - If the model is inactive (disabled or PAL), this is a no-op — `oam` and
2039    ///   the timestamps are left untouched, so the read is byte-identical to stock.
2040    /// - Else, for the 8-byte row containing `addr`: if the last touch was within
2041    ///   [`OAM_DECAY_CPU_CYCLES`] CPU cycles, refresh the row's timestamp (the DRAM
2042    ///   cell was recharged by this access). Otherwise the row has decayed — rewrite
2043    ///   all 8 of its bytes to the canonical pattern `((sprAddr & 3) == 2) ?
2044    ///   (sprAddr & 0xE3) : sprAddr` (the attribute byte keeps only its implemented
2045    ///   bits; the others read back their own low address) and leave the stale
2046    ///   timestamp (so the row keeps reading decayed until a write refreshes it,
2047    ///   matching the documented decay behavior).
2048    ///
2049    /// The subsequent `oam[addr]` read then returns the (possibly decayed) byte.
2050    #[inline]
2051    fn oam_decay_on_read(&mut self, addr: u8) {
2052        if !self.oam_decay_active() {
2053            return;
2054        }
2055        let row = (addr >> 3) as usize;
2056        let now = self.dot_counter / 3;
2057        // Saturating (wrapping) subtraction: `now` is monotone ≥ the stored
2058        // timestamp in practice, but `wrapping_sub` keeps this total even across a
2059        // (astronomically unlikely) u64 counter wrap.
2060        let elapsed = now.wrapping_sub(self.oam_decay_cycles[row]);
2061        if elapsed <= OAM_DECAY_CPU_CYCLES {
2062            self.oam_decay_cycles[row] = now;
2063        } else {
2064            let base = addr & 0xF8;
2065            for i in 0..8u8 {
2066                let spr_addr = base | i;
2067                self.oam[spr_addr as usize] = if spr_addr & 0x03 == 0x02 {
2068                    spr_addr & 0xE3
2069                } else {
2070                    spr_addr
2071                };
2072            }
2073        }
2074    }
2075
2076    /// v2.1.4 F2.3 — OAM-write decay hook. Call **after** writing `oam[addr]` at
2077    /// every primary-OAM write site (`$2004` / OAM DMA). Implements the documented
2078    /// OAM DRAM decay-on-write refresh (`NESdev` wiki "PPU OAM"): a write recharges
2079    /// the row's DRAM cells, so refresh the
2080    /// row's last-touch timestamp. Inactive (disabled or PAL) ⇒ no-op, so the write
2081    /// path is byte-identical to stock at the default.
2082    #[inline]
2083    const fn oam_decay_on_write(&mut self, addr: u8) {
2084        if !self.oam_decay_active() {
2085            return;
2086        }
2087        self.oam_decay_cycles[(addr >> 3) as usize] = self.dot_counter / 3;
2088    }
2089
2090    /// Rebuild [`Self::rgba_lut`] from the custom palette when one is loaded,
2091    /// otherwise from the active built-in [`crate::palette::PpuPalette`].
2092    const fn rebuild_rgba_lut(&mut self) {
2093        self.rgba_lut = match &self.custom_palette {
2094            Some(base) => build_rgba_lut_from_base(base),
2095            None => build_rgba_lut(self.active_palette),
2096        };
2097    }
2098
2099    /// Map a CPU-visible PPU register index (0-7) to the internal register,
2100    /// applying the 2C05 `$2000`<->`$2001` swap.
2101    ///
2102    /// On a 2C05 a write/read of `$2000` (reg 0) targets MASK (reg 1) and vice
2103    /// versa; all other registers are unaffected. On every other PPU (the
2104    /// default path) this is the identity, so normal NES behaviour is unchanged.
2105    const fn map_register(&self, reg: u8) -> u8 {
2106        if self.is_2c05 {
2107            match reg & 7 {
2108                0 => 1,
2109                1 => 0,
2110                other => other,
2111            }
2112        } else {
2113            reg & 7
2114        }
2115    }
2116
2117    /// Returns a reference to the internal CIRAM (nametables).
2118    pub fn vram_ref(&self) -> &[u8] {
2119        &self.ciram
2120    }
2121
2122    /// Returns a mutable reference to the internal CIRAM (nametables).
2123    pub fn vram_mut(&mut self) -> &mut [u8] {
2124        &mut self.ciram
2125    }
2126
2127    /// Performs a soft-reset of the PPU (warm boot). Per `docs/ppu-2c02.md`:
2128    ///   - PPUCTRL := 0
2129    ///   - PPUMASK := 0
2130    ///   - w toggle := 0
2131    ///   - PPUSTATUS bits 7 (VBL) unchanged on real hardware (we leave it
2132    ///     as-is for parity with `$2002`-race tests)
2133    ///   - PPUDATA buffer := 0
2134    ///   - Mask window restarts (writes to $2000/$2001/$2005/$2006 ignored
2135    ///     for the documented number of cycles after reset).
2136    pub const fn reset(&mut self) {
2137        self.ctrl = PpuCtrl::empty();
2138        self.mask = PpuMask::empty();
2139        self.mask_for_skip_check = PpuMask::empty();
2140        self.mask_skip_pipe1 = PpuMask::empty();
2141        self.prev_rendering_enabled = false;
2142        self.rendering_enabled_delayed = false;
2143        self.rendering_enabled_delayed2 = false;
2144        self.spr_rearm_deferred = false;
2145        // The rendering-gate pipeline is the same class of state as the two
2146        // skip-check stages above and was simply missed. Reset preserves the
2147        // current dot, so a reset at dot 254 otherwise reaches dot 256 with
2148        // stale history and fires `inc_vert_v()` against an empty PPUMASK.
2149        // Found in review on #515; the two-dot stage widened the window that
2150        // made it observable, but `rendering_enabled_delayed` had the same hole.
2151        self.w = false;
2152        self.data_buffer = 0;
2153        self.post_reset_mask_remaining = self.region.post_reset_mask_cycles();
2154        self.nmi_line = false;
2155        // v2.1.4 F2.3 — mark every OAM row freshly refreshed at reset, matching
2156        // Mesen2's `NesPpu::Reset` (which stamps `_oamDecayCycles` with the current
2157        // clock unconditionally). Doing this regardless of the enable flag keeps the
2158        // timestamps sane if decay is toggled on after a reset, and is inert while
2159        // decay is off (the array is never read). `dot_counter / 3` is the current
2160        // CPU cycle (NTSC/Dendy have 3 dots per CPU cycle).
2161        let now = self.dot_counter / 3;
2162        let mut i = 0;
2163        while i < self.oam_decay_cycles.len() {
2164            self.oam_decay_cycles[i] = now;
2165            i += 1;
2166        }
2167    }
2168
2169    /// Returns `true` if the PPU is asserting the NMI line.
2170    #[must_use]
2171    pub const fn nmi_line(&self) -> bool {
2172        self.nmi_line
2173    }
2174
2175    /// Consume and return the per-frame "frame complete" latch.
2176    pub const fn take_frame_complete(&mut self) -> bool {
2177        let r = self.frame_complete;
2178        self.frame_complete = false;
2179        r
2180    }
2181
2182    /// Install a state-trace buffer. Subsequent calls to
2183    /// [`Self::tick`] will append one [`PpuStateRecord`] per dot
2184    /// for dots inside the buffer's filter window. Pre-existing
2185    /// records (if any) are dropped.
2186    ///
2187    /// Read-only: every call to [`Self::tick`] reads PPU state
2188    /// after the dot's effects have applied; it never mutates
2189    /// emulator state, so the determinism contract is preserved
2190    /// (`docs/architecture.md` §Determinism).
2191    ///
2192    /// See `docs/adr/0005-ppu-state-trace.md` and the rustdoc on
2193    /// [`crate::state_trace`].
2194    ///
2195    /// [`PpuStateRecord`]: crate::state_trace::PpuStateRecord
2196    #[cfg(feature = "ppu-state-trace")]
2197    pub fn enable_state_trace(&mut self, trace: crate::state_trace::PpuStateTrace) {
2198        self.state_trace = Some(trace);
2199    }
2200
2201    /// Stamp the CPU cycle that the dots ticked next belong to.
2202    ///
2203    /// Called by the bus at the START of each CPU cycle, so a record carries
2204    /// the number of the cycle it is part of rather than the following one —
2205    /// the off-by-one a co-simulation probe made on exactly this question, and
2206    /// which produced a finding that had to be retracted.
2207    #[cfg(feature = "ppu-state-trace")]
2208    pub const fn set_trace_cpu_cycle(&mut self, cycle: u64) {
2209        self.trace_cpu_cycle = cycle;
2210    }
2211
2212    /// Install a per-dot PPU bus address capture.
2213    ///
2214    /// The address bus is pin-observable, which is what makes it usable as a
2215    /// gate by an independent reimplementation; see [`crate::fetch_trace`] for
2216    /// why the address is captured rather than derived.
2217    #[cfg(feature = "ppu-fetch-trace")]
2218    pub fn enable_fetch_trace(&mut self, trace: crate::fetch_trace::FetchTrace) {
2219        self.fetch_trace = Some(trace);
2220    }
2221
2222    /// Remove and return the fetch trace, if one was installed.
2223    #[cfg(feature = "ppu-fetch-trace")]
2224    pub const fn take_fetch_trace(&mut self) -> Option<crate::fetch_trace::FetchTrace> {
2225        self.fetch_trace.take()
2226    }
2227
2228    /// Take the accumulated state trace, leaving the PPU's trace
2229    /// slot empty. Returns `None` if tracing was never enabled.
2230    #[cfg(feature = "ppu-state-trace")]
2231    #[must_use]
2232    pub const fn take_state_trace(&mut self) -> Option<crate::state_trace::PpuStateTrace> {
2233        self.state_trace.take()
2234    }
2235
2236    /// Borrow the in-flight state trace without taking it.
2237    #[cfg(feature = "ppu-state-trace")]
2238    #[must_use]
2239    pub const fn state_trace(&self) -> Option<&crate::state_trace::PpuStateTrace> {
2240        self.state_trace.as_ref()
2241    }
2242
2243    /// Build a [`PpuStateRecord`] snapshot from the PPU's
2244    /// current state. Used by the per-dot recording hook at the
2245    /// end of [`Self::tick`]; exposed publicly so external
2246    /// tooling (e.g. the trace fixture's end-of-frame snapshot)
2247    /// can re-use the canonical packer.
2248    ///
2249    /// [`PpuStateRecord`]: crate::state_trace::PpuStateRecord
2250    #[cfg(feature = "ppu-state-trace")]
2251    #[must_use]
2252    pub fn build_state_record(&self) -> crate::state_trace::PpuStateRecord {
2253        crate::state_trace::PpuStateRecord {
2254            // Frames easily exceed u16 over a 600-frame test run.
2255            // The `as u32` truncates the upper bits of the u64
2256            // counter — which is fine for any realistic capture
2257            // window (u32::MAX ≈ 71 days of NES wall time).
2258            frame: self.frame as u32,
2259            scanline: self.scanline,
2260            dot: self.dot,
2261            ctrl: self.ctrl.bits(),
2262            mask: self.mask.bits(),
2263            status: self.status.bits(),
2264            oam_addr: self.oam_addr,
2265            v: self.v,
2266            t: self.t,
2267            fine_x: self.x,
2268            w_toggle: self.w,
2269            sprite_eval_n: self.sprite_eval_n,
2270            sprite_eval_m: self.sprite_eval_m,
2271            sprite_eval_found: self.sprite_eval_found,
2272            sprite_eval_sec_idx: self.sprite_eval_sec_idx,
2273            sprite_eval_copying: self.sprite_eval_copying,
2274            sprite_eval_overflow_search: self.sprite_eval_overflow_search,
2275            sprite_eval_done: self.sprite_eval_done,
2276            sprite_eval_read_latch: self.sprite_eval_read_latch,
2277            spr_count: self.spr_count,
2278            spr_zero_in_line: self.spr_zero_in_line,
2279            spr_shift_lo: self.spr_shift_lo,
2280            spr_shift_hi: self.spr_shift_hi,
2281            spr_attr: self.spr_attr,
2282            spr_x: self.spr_x,
2283            bg_shift_lo: self.bg_shift_lo,
2284            bg_shift_hi: self.bg_shift_hi,
2285            at_shift_lo: self.at_shift_lo,
2286            at_shift_hi: self.at_shift_hi,
2287            nt_latch: self.nt_latch,
2288            at_latch: self.at_latch,
2289            bg_lo_latch: self.bg_lo_latch,
2290            bg_hi_latch: self.bg_hi_latch,
2291            secondary_oam: self.secondary_oam,
2292            oam_fnv1a64: crate::state_trace::fnv1a64(&self.oam),
2293            nmi_line: self.nmi_line,
2294            oam_bus_copybuffer: self.oam_data_bus_observed(),
2295            data_buffer: self.data_buffer,
2296            cpu_cycle: self.trace_cpu_cycle,
2297        }
2298    }
2299
2300    /// Borrow the (possibly partial) framebuffer.
2301    #[must_use]
2302    pub fn framebuffer(&self) -> &[u8] {
2303        &self.framebuffer
2304    }
2305
2306    /// v1.7.0 "Forge" Workstream B (B3) — overwrite the RGBA8 output framebuffer
2307    /// (the Lua `emu:setScreenBuffer(t)` paints output only). Copies up to the
2308    /// framebuffer length; a short source leaves the tail untouched. Output-only
2309    /// — it touches only the display buffer the frontend presents, NOT any
2310    /// register / latch / scroll state, so the determinism contract is
2311    /// unaffected (a later real frame fully repaints it). `debug-hooks`-gated and
2312    /// reached only through the script crate's gated post-frame path, so the
2313    /// shipped build is byte-identical.
2314    #[cfg(feature = "debug-hooks")]
2315    pub fn debug_set_framebuffer(&mut self, rgba: &[u8]) {
2316        let n = rgba.len().min(self.framebuffer.len());
2317        self.framebuffer[..n].copy_from_slice(&rgba[..n]);
2318    }
2319
2320    /// Borrow the parallel per-pixel **palette-index** framebuffer
2321    /// (256 × 240 `u16`s, each `(emphasis << 6) | colour`, 0..=511) used by the
2322    /// true composite `NES_NTSC` filter (T-110-A1). A faithful index-space mirror
2323    /// of [`Self::framebuffer`]; output-only, so the determinism contract holds.
2324    #[must_use]
2325    pub fn index_framebuffer(&self) -> &[u16] {
2326        &self.index_framebuffer
2327    }
2328
2329    /// Pack a background tile's active palette into Mesen's `PaletteColors` key
2330    /// form (`pr[base+3] | pr[base+2]<<8 | pr[base+1]<<16 | pr[0]<<24`, with
2331    /// `base = $3F00 | group<<2` and `pr[0]` the universal backdrop). Used only
2332    /// to key HD-pack tile replacements; output-only.
2333    #[cfg(feature = "hd-pack")]
2334    fn hd_bg_palette_colors(&self, group: u8) -> u32 {
2335        let base = 0x3F00 | (u16::from(group) << 2);
2336        let p0 = u32::from(self.read_palette(0x3F00) & 0x3F);
2337        let p1 = u32::from(self.read_palette(base | 1) & 0x3F);
2338        let p2 = u32::from(self.read_palette(base | 2) & 0x3F);
2339        let p3 = u32::from(self.read_palette(base | 3) & 0x3F);
2340        p3 | (p2 << 8) | (p1 << 16) | (p0 << 24)
2341    }
2342
2343    /// Pack a sprite tile's palette (`0xFF000000 | pr[base+3] | pr[base+2]<<8 |
2344    /// pr[base+1]<<16`, `base = $3F10 | group<<2`; the `0xFF` top byte is the
2345    /// sprite/BG discriminator and there is no `pr[0]` term).
2346    #[cfg(feature = "hd-pack")]
2347    fn hd_sprite_palette_colors(&self, group: u8) -> u32 {
2348        let base = 0x3F10 | (u16::from(group) << 2);
2349        let p1 = u32::from(self.read_palette(base | 1) & 0x3F);
2350        let p2 = u32::from(self.read_palette(base | 2) & 0x3F);
2351        let p3 = u32::from(self.read_palette(base | 3) & 0x3F);
2352        0xFF00_0000 | p3 | (p2 << 8) | (p1 << 16)
2353    }
2354
2355    /// v1.2.0 beta.2 (Workstream C3) — borrow the per-pixel HD-pack
2356    /// tile-source buffer (256 × 240 [`HdTileSource`] records, parallel to
2357    /// [`Self::index_framebuffer`]). Each entry names the CHR tile that
2358    /// produced the pixel. Output-only telemetry; the determinism /
2359    /// `AccuracyCoin` contract is unaffected. See
2360    /// `docs/ppu-2c02.md` §HD-pack tile-source export.
2361    #[cfg(feature = "hd-pack")]
2362    #[must_use]
2363    pub fn hd_tile_source(&self) -> &[HdTileSource] {
2364        &self.hd_tile_source
2365    }
2366
2367    /// The per-frame NTSC composite colour phase — the `videoPhase` the
2368    /// `NES_NTSC` filter feeds its signal generator. `0..=2` on NTSC; on
2369    /// PAL/Dendy it is the frame parity (`0..=1`). Snapshotted at the last frame
2370    /// boundary. Cosmetic (drives only the optional filter's dot-crawl).
2371    #[must_use]
2372    pub const fn ntsc_phase(&self) -> u8 {
2373        self.frame_ntsc_phase
2374    }
2375
2376    /// Snapshot the per-frame NTSC colour phase from the master-cycle counter.
2377    /// NES NTSC steps the colour phase through 3 frame states (the source of the
2378    /// dot-crawl); PAL/Dendy have no equivalent 3-phase crawl, so the frame
2379    /// parity is exposed instead. Called at each frame boundary.
2380    const fn snapshot_ntsc_phase(&mut self) {
2381        self.frame_ntsc_phase = if matches!(self.region, PpuRegion::Ntsc) {
2382            (self.dot_counter % 3) as u8
2383        } else {
2384            (self.frame & 1) as u8
2385        };
2386    }
2387
2388    /// Current dot (0..=340).
2389    #[must_use]
2390    pub const fn dot(&self) -> u16 {
2391        self.dot
2392    }
2393
2394    /// Current scanline.
2395    #[must_use]
2396    pub const fn scanline(&self) -> i16 {
2397        self.scanline
2398    }
2399
2400    /// Current frame counter.
2401    #[must_use]
2402    pub const fn frame(&self) -> u64 {
2403        self.frame
2404    }
2405
2406    /// Snapshot of CPU-visible register bytes (for the debugger UI).
2407    ///
2408    /// Returns `[ctrl, mask, status, oam_addr]`. Read-only — does NOT clear
2409    /// VBL or toggle the write latch (unlike `cpu_read_register`).
2410    #[must_use]
2411    pub const fn debug_registers(&self) -> [u8; 4] {
2412        [
2413            self.ctrl.bits(),
2414            self.mask.bits(),
2415            self.status.bits(),
2416            self.oam_addr,
2417        ]
2418    }
2419
2420    /// Snapshot of loopy scroll registers `(v, t, x, w)`.
2421    #[must_use]
2422    pub const fn debug_scroll(&self) -> (u16, u16, u8, bool) {
2423        (self.v, self.t, self.x, self.w)
2424    }
2425
2426    /// v1.8.9 — the frame's background scroll `(x, y)` in NES pixels, decoded
2427    /// from the `t` (temp VRAM addr) register + fine-X, including the nametable
2428    /// bits (Mesen HD-pack `_scrollX`/`scrollY`). Used by the HD compositor to
2429    /// offset parallax `<background>` layers by `scroll * ratio`. A frame-level
2430    /// value (the scroll at `t`), not per-scanline. Output-only.
2431    #[must_use]
2432    pub const fn hd_bg_scroll(&self) -> (i32, i32) {
2433        let t = self.t;
2434        let x = ((t & 0x1F) << 3) | (self.x as u16) | if t & 0x0400 != 0 { 0x100 } else { 0 };
2435        let y =
2436            (((t & 0x03E0) >> 2) | ((t & 0x7000) >> 12)) + if t & 0x0800 != 0 { 240 } else { 0 };
2437        (x as i32, y as i32)
2438    }
2439
2440    /// Borrow the 32-byte palette RAM (read-only).
2441    #[must_use]
2442    pub const fn palette_ram(&self) -> &[u8; 32] {
2443        &self.palette_ram
2444    }
2445
2446    /// Borrow OAM (256 bytes = 64 sprites x 4 bytes).
2447    #[must_use]
2448    pub fn oam(&self) -> &[u8] {
2449        &self.oam
2450    }
2451
2452    /// Borrow nametable CIRAM (2 KiB).
2453    #[must_use]
2454    pub fn ciram(&self) -> &[u8] {
2455        &self.ciram
2456    }
2457
2458    /// v2.3.2 "Lucid" — arm or disarm per-byte write attribution.
2459    ///
2460    /// Arming allocates [`crate::provenance::WriteAttribution::HEAP_BYTES`] and
2461    /// starts stamping every subsequent CIRAM / OAM / palette write with the
2462    /// writing instruction's PC and cycle. Disarming frees the store outright, so
2463    /// re-arming starts from a clean slate rather than resurrecting stale records
2464    /// from a previous debugging session.
2465    ///
2466    /// Purely observational — nothing in the render or timing path reads it, so
2467    /// output is bit-identical either way.
2468    #[cfg(feature = "debug-hooks")]
2469    pub fn set_write_attribution(&mut self, enabled: bool) {
2470        self.write_attrib = if enabled {
2471            Some(Box::new(crate::provenance::WriteAttribution::new()))
2472        } else {
2473            None
2474        };
2475    }
2476
2477    /// The write-attribution store, or `None` when not armed.
2478    #[cfg(feature = "debug-hooks")]
2479    #[must_use]
2480    pub fn write_attribution(&self) -> Option<&crate::provenance::WriteAttribution> {
2481        self.write_attrib.as_deref()
2482    }
2483
2484    /// Forget every recorded attribution, keeping the store armed.
2485    ///
2486    /// The core calls this on power-cycle and on save-state restore: the restored
2487    /// bytes were not written by any instruction this session ran, and reporting
2488    /// the PCs that happened to write those offsets *before* the restore would be
2489    /// a confidently wrong answer rather than an absent one.
2490    #[cfg(feature = "debug-hooks")]
2491    pub fn clear_write_attribution(&mut self) {
2492        if let Some(attrib) = self.write_attrib.as_mut() {
2493            attrib.clear();
2494        }
2495    }
2496
2497    /// Push down the `(pc, cycle)` of the CPU instruction whose write is about to
2498    /// land, so the store site can stamp it. Called by the bus immediately before
2499    /// a `$2000-$3FFF` register write and before an OAM DMA burst.
2500    ///
2501    /// A no-op when attribution is not armed, and never read by emulation.
2502    #[cfg(feature = "debug-hooks")]
2503    pub const fn set_attrib_context(&mut self, pc: u16, cycle: u64) {
2504        self.attrib_pc = pc;
2505        self.attrib_cycle = cycle;
2506    }
2507
2508    /// v2.3.2 "Lucid" phase 2 — arm or disarm per-pixel provenance capture.
2509    ///
2510    /// Arming allocates
2511    /// [`crate::provenance::PixelProvenanceFrame::HEAP_BYTES`] and starts
2512    /// recording, for every emitted pixel, the layer that won, the exact palette
2513    /// address, and the nametable / attribute / pattern addresses of the tile
2514    /// actually on screen. Disarming frees the frame.
2515    ///
2516    /// Independent of [`Self::set_write_attribution`]: this says *which bytes*
2517    /// produced a pixel, that says *who wrote* those bytes. The panel wants
2518    /// both, but each is useful alone and neither depends on the other.
2519    ///
2520    /// Output-only, so emulation is bit-identical either way.
2521    #[cfg(feature = "debug-hooks")]
2522    pub fn set_pixel_provenance(&mut self, enabled: bool) {
2523        self.prov_frame = if enabled {
2524            Some(Box::new(crate::provenance::PixelProvenanceFrame::new()))
2525        } else {
2526            None
2527        };
2528        self.prov_armed = enabled;
2529    }
2530
2531    /// The current frame's per-pixel provenance, or `None` when not armed.
2532    #[cfg(feature = "debug-hooks")]
2533    #[must_use]
2534    pub fn pixel_provenance(&self) -> Option<&crate::provenance::PixelProvenanceFrame> {
2535        self.prov_frame.as_deref()
2536    }
2537
2538    /// Forget every recorded pixel, keeping the frame armed.
2539    ///
2540    /// Mirrors [`Self::clear_write_attribution`], and for the same reason: a
2541    /// restore lands mid-frame, so without this the panel would report tile and
2542    /// palette addresses from the abandoned timeline for every pixel above the
2543    /// current scanline, with nothing marking them stale.
2544    #[cfg(feature = "debug-hooks")]
2545    pub fn clear_pixel_provenance(&mut self) {
2546        if let Some(frame) = self.prov_frame.as_mut() {
2547            frame.clear();
2548        }
2549    }
2550
2551    /// Move both provenance stores out, leaving the PPU unarmed.
2552    ///
2553    /// Paired with [`Self::put_provenance`] to carry the stores across a
2554    /// same-timeline restore that would otherwise clear them — see
2555    /// [`crate::provenance::ProvenanceStash`] for why run-ahead needs that and
2556    /// save-state loads and netplay rollback do not.
2557    ///
2558    /// `prov_armed` is dropped to `false` alongside the frame it mirrors, so the
2559    /// invariant "`prov_armed` iff `prov_frame.is_some()`" holds while stashed
2560    /// and `emit_pixel` records nothing into the vacated slot.
2561    #[cfg(feature = "debug-hooks")]
2562    pub const fn take_provenance(&mut self) -> crate::provenance::ProvenanceStash {
2563        let stash = crate::provenance::ProvenanceStash {
2564            write_attrib: self.write_attrib.take(),
2565            prov_frame: self.prov_frame.take(),
2566            prov_armed: self.prov_armed,
2567        };
2568        self.prov_armed = false;
2569        stash
2570    }
2571
2572    /// Put back stores taken by [`Self::take_provenance`].
2573    ///
2574    /// Overwrites whatever is currently held, which is what the pairing wants:
2575    /// the only thing that can have appeared in between is a restore's cleared
2576    /// (or absent) store, and the stashed records are the ones the caller means
2577    /// to keep.
2578    #[cfg(feature = "debug-hooks")]
2579    pub fn put_provenance(&mut self, stash: crate::provenance::ProvenanceStash) {
2580        self.write_attrib = stash.write_attrib;
2581        self.prov_frame = stash.prov_frame;
2582        self.prov_armed = stash.prov_armed;
2583    }
2584
2585    /// Freeze the current instruction context as the cause of an OAM DMA burst.
2586    ///
2587    /// Called by the bus from the `$4014` write, i.e. while
2588    /// [`Self::set_attrib_context`] still holds the `STA $4014` itself.
2589    ///
2590    /// The burst cannot use the live context: `$4014` only arms the transfer, and
2591    /// its 513 or 514 cycles are then stolen from the instructions that follow,
2592    /// so by the time the first OAM byte lands the live context names whichever
2593    /// instruction is being halted — true about the timing, wrong about the cause.
2594    #[cfg(feature = "debug-hooks")]
2595    pub const fn latch_dma_attrib_context(&mut self) {
2596        self.dma_attrib_pc = self.attrib_pc;
2597        self.dma_attrib_cycle = self.attrib_cycle;
2598    }
2599
2600    /// v1.7.0 "Forge" Workstream A1 — debugger writeback: store one palette-RAM
2601    /// byte directly (`idx` masked to 0..32, value masked to the 6-bit palette
2602    /// width), reusing the same canonical mirroring/masking as the live
2603    /// `$2007` write path. Used only by the `debug-hooks` editor writeback,
2604    /// which routes through the gated post-frame poke path — so the default
2605    /// (no-edit) build never calls it and stays byte-identical.
2606    #[cfg(feature = "debug-hooks")]
2607    pub const fn debug_poke_palette(&mut self, idx: u8, value: u8) {
2608        // `palette_index` mirrors $3F10/$14/$18/$1C → $3F00/.. and folds the
2609        // 32-byte window; feed it the raw address so an editor index maps the
2610        // same way a $2007 write would.
2611        let addr = 0x3F00u16 | ((idx & 0x1F) as u16);
2612        let i = palette_index(addr);
2613        self.palette_ram[i] = value & 0x3F;
2614    }
2615
2616    /// v1.7.0 "Forge" Workstream A1 — debugger writeback: store one OAM byte
2617    /// directly. `debug-hooks`-gated; only reached through the gated post-frame
2618    /// poke path, so the default build is byte-identical.
2619    #[cfg(feature = "debug-hooks")]
2620    pub const fn debug_poke_oam(&mut self, idx: u8, value: u8) {
2621        self.oam[idx as usize] = value;
2622    }
2623
2624    /// v1.7.0 "Forge" Workstream A1 — debugger writeback: store one CIRAM byte
2625    /// at a physical offset (caller resolves mirroring via the mapper).
2626    /// `debug-hooks`-gated; only reached through the gated post-frame poke path.
2627    #[cfg(feature = "debug-hooks")]
2628    pub const fn debug_poke_ciram(&mut self, phys: usize, value: u8) {
2629        self.ciram[phys & 0x07FF] = value;
2630    }
2631
2632    /// `true` when sprites are rendered in 8x16 mode (CTRL bit 5).
2633    #[must_use]
2634    pub const fn sprite_size_16(&self) -> bool {
2635        self.ctrl
2636            .contains(crate::registers::PpuCtrl::SPRITE_SIZE_16)
2637    }
2638
2639    /// Base address of the BG pattern table (`$0000` or `$1000`).
2640    #[must_use]
2641    pub const fn bg_pattern_base(&self) -> u16 {
2642        if self
2643            .ctrl
2644            .contains(crate::registers::PpuCtrl::BG_PATTERN_HIGH)
2645        {
2646            0x1000
2647        } else {
2648            0x0000
2649        }
2650    }
2651
2652    /// Base address of the sprite pattern table (8x8 mode only).
2653    #[must_use]
2654    pub const fn sprite_pattern_base(&self) -> u16 {
2655        if self
2656            .ctrl
2657            .contains(crate::registers::PpuCtrl::SPRITE_PATTERN_HIGH)
2658        {
2659            0x1000
2660        } else {
2661            0x0000
2662        }
2663    }
2664
2665    /// OAM DMA byte write: place `value` at `oam[oam_addr]` and increment
2666    /// `oam_addr`. Used by the bus's OAM DMA state machine.
2667    ///
2668    /// Bypasses the OAMADDR-during-rendering corruption modeled by
2669    /// `cpu_write_register` for `$2004` direct writes — DMA writes always
2670    /// hit OAM directly per nesdev.
2671    pub fn oam_dma_write(&mut self, value: u8) {
2672        // v2.9.5 (sibling ledger 3.1c): a DMA byte is a `$2004` write, and
2673        // "writing any value to any PPU port ... will fill this latch"
2674        // (`nesdev_wiki/PPU_registers`). The MiSTer core did this all along.
2675        self.touch_open_bus(value);
2676        self.oam[self.oam_addr as usize] = value;
2677        // v2.3.2 "Lucid" — every byte of the burst is attributed to the ONE
2678        // `STA $4014` that triggered it (the LATCHED context, not the live one:
2679        // the burst steals cycles from the instructions AFTER the trigger, so
2680        // the live context names the halted instruction rather than the cause).
2681        // 256 bytes genuinely share one cause, and reporting anything else would
2682        // invent a history the program does not have.
2683        #[cfg(feature = "debug-hooks")]
2684        if let Some(attrib) = self.write_attrib.as_mut() {
2685            attrib.record_oam(
2686                self.oam_addr,
2687                self.dma_attrib_pc,
2688                self.dma_attrib_cycle,
2689                value,
2690            );
2691        }
2692        // v2.1.4 F2.3 — OAM-decay write hook (no-op at the default): the DMA byte
2693        // recharges the written row's DRAM cells, so refresh its timestamp. Mesen2
2694        // routes DMA writes through the same `WriteSpriteRam` refresh.
2695        self.oam_decay_on_write(self.oam_addr);
2696        self.oam_addr = self.oam_addr.wrapping_add(1);
2697    }
2698
2699    /// Notify the PPU that one CPU cycle has elapsed. Used to drive the
2700    /// post-reset masking window and the open-bus decay timers.
2701    pub const fn on_cpu_cycle(&mut self) {
2702        self.post_reset_mask_remaining = self.post_reset_mask_remaining.saturating_sub(1);
2703        // Open-bus decay: per-bit-group, three independent timers.  When a
2704        // group's timer hits 0 those bits clear in the latch.  Per
2705        // docs/ppu-2c02.md, real hardware decays in 3-30 ms; we use
2706        // **558.7 ms** — one million CPU cycles at NTSC.
2707        //
2708        // The figure in this comment used to read "~600 ms (≈ 1,073,447 CPU
2709        // cycles at NTSC, rounded to one million)".  The arithmetic in the
2710        // parenthesis is right and the headline is not: one million cycles is
2711        // 558.7 ms, so "rounded" was a 7% cut, not a rounding.  Corrected here
2712        // because the MiSTer co-simulation DUT has to reproduce this number
2713        // exactly and was quoting the wrong one back.
2714        //
2715        // SWEPT (v2.6.3), because "conservative but well within the window the
2716        // `ppu_open_bus` test cares about" turned out to understate how much
2717        // slack there is.  That ROM's decay checks are 100 × `delay_msec 10`
2718        // loops asserting the value has reached zero, so its only real
2719        // constraint is < 1000 ms — and it passes at **3, 10, 20, 30 and
2720        // 100 ms** as well.  AccuracyCoin holds 141/141 (RAM decoder), nestest
2721        // matches its golden log, and all eight `nes_blargg` tests pass at
2722        // 30 ms, the documented upper bound.
2723        //
2724        // So this constant is NOT forced by the corpus: a documentation-derived
2725        // value is measurably available.  It is kept at one million because
2726        // changing it changes shipped emulator behaviour for every game that
2727        // reads open bus after a long gap, and no test in this repository can
2728        // adjudicate which is right — the wiki's band and this model differ by
2729        // ~19×, and neither has an independent oracle here.  Recorded as a
2730        // deliberate hold rather than a derivation.  See
2731        // `RustyNES_MiSTer/docs/rung3-ppu.md`, where the same constant is
2732        // reproduced as 3,000,000 dots and the DUT-side sweep is written up.
2733        // NOTE (v2.3.1 G5): reformulating this as a deadline comparison instead
2734        // of a per-cycle decrement was measured by DELETING the loop outright —
2735        // the ceiling any reformulation could reach — and the ceiling is ZERO.
2736        // ~29,780 calls/frame sounds expensive; it is three predictable
2737        // compare-and-decrement steps on data already in L1, which an
2738        // out-of-order core absorbs entirely. Do not re-attempt; see
2739        // `docs/performance.md`.
2740        let mut i = 0;
2741        while i < 3 {
2742            if self.open_bus_decay[i] > 0 {
2743                self.open_bus_decay[i] -= 1;
2744                if self.open_bus_decay[i] == 0 {
2745                    self.open_bus &= !Self::OPEN_BUS_GROUP_MASKS[i];
2746                }
2747            }
2748            i += 1;
2749        }
2750    }
2751
2752    /// Per-bit-group masks for the open-bus latch decay model.  Group 0 is
2753    /// bits 0-4 (refreshed by writes, $2004 reads, and $2007 reads — both
2754    /// palette and non-palette).  Group 1 is bit 5 (refreshed by writes,
2755    /// $2002 reads, $2004 reads, and $2007 reads).  Group 2 is bits 6-7
2756    /// (refreshed by writes, $2002 reads, $2004 reads, and $2007 non-palette
2757    /// reads — but not by palette reads).
2758    const OPEN_BUS_GROUP_MASKS: [u8; 3] = [0x1F, 0x20, 0xC0];
2759
2760    /// Decay-timer reload value (~600 ms at NTSC).
2761    const OPEN_BUS_DECAY_RELOAD: u32 = 1_000_000;
2762
2763    /// Refresh the open-bus latch.  `group_mask` is a bitmap selecting which
2764    /// of the three decay groups to refresh: bit 0 = bits 0-4, bit 1 = bit 5,
2765    /// bit 2 = bits 6-7.  Only the bits in those groups are copied from
2766    /// `value`; bits in groups not selected retain their previous latch value
2767    /// and their decay timer is left to keep counting down.
2768    const fn refresh_open_bus(&mut self, value: u8, group_mask: u8) {
2769        let mut i = 0;
2770        while i < 3 {
2771            if (group_mask >> i) & 1 == 1 {
2772                let m = Self::OPEN_BUS_GROUP_MASKS[i];
2773                self.open_bus = (self.open_bus & !m) | (value & m);
2774                self.open_bus_decay[i] = Self::OPEN_BUS_DECAY_RELOAD;
2775            }
2776            i += 1;
2777        }
2778    }
2779
2780    /// Refresh **all** bit groups of the open-bus latch — used by writes and
2781    /// any read that drives all 8 bits (e.g. $2004 OAMDATA, $2007 non-palette
2782    /// PPUDATA).
2783    const fn touch_open_bus(&mut self, value: u8) {
2784        self.refresh_open_bus(value, 0b111);
2785    }
2786
2787    /// Number of PPU dots between a `$2002` read's start (M2 high, when the
2788    /// VBL flag is latched) and its end (M2 low, when the *unlatched*
2789    /// sprite-0-hit / overflow flags are sampled). On a revision-G 2A03 M2's
2790    /// 15/24 duty cycle puts read-end ~1.875 PPU dots after read-start; we
2791    /// round to the nearest whole dot for the lockstep computed sample. This
2792    /// is the empirically-tuned knob for the `$2002 flag timing` test.
2793    ///
2794    /// Tuned to **1**: with the test's reads spaced 1 PPU dot apart, a 1-dot
2795    /// window yields the primary answer key `$E0,$E0,$80,$00` (read 3 = `$80`:
2796    /// VBL latched set, sprite flags read 0). A 2-dot window also passes Test 1
2797    /// (via the alt key `$E0,$80,$80,$00`) but masks one extra read position,
2798    /// which regresses `$2004 Stress` (whose `$2002` sync read lands there).
2799    const STATUS_READ_END_DOTS: u16 = 1;
2800
2801    /// True when advancing `dots` PPU dots from the current `(scanline, dot)`
2802    /// reaches or passes the pre-render dot-1 flag-clear, where the
2803    /// sprite-0-hit and overflow flags are cleared. Used by the `$2002`
2804    /// two-point read model to sample bits 6/5 as-of read-end. Forward
2805    /// distance only: a position already at/after this frame's clear is *not*
2806    /// "imminent" (its flags are already cleared in the status register, and
2807    /// the full-frame wrap distance never satisfies the small `dots` bound).
2808    fn sprite_flags_clear_imminent(&self, dots: u16) -> bool {
2809        const DOTS_PER_LINE: i32 = 341;
2810        let lines = i32::from(self.region.prerender_line()) + 1;
2811        let frame_dots = lines * DOTS_PER_LINE;
2812        let target = i32::from(self.region.prerender_line()) * DOTS_PER_LINE + 1;
2813        let cur = i32::from(self.scanline) * DOTS_PER_LINE + i32::from(self.dot);
2814        let until = (target - cur).rem_euclid(frame_dots);
2815        until > 0 && until <= i32::from(dots)
2816    }
2817
2818    /// CPU register read at `$2000-$3FFF` (only the low 3 bits matter).
2819    #[allow(clippy::too_many_lines)]
2820    pub fn cpu_read_register<B: PpuBus>(&mut self, reg: u8, bus: &mut B) -> u8 {
2821        // 2C05 swaps $2000<->$2001 (no-op on every other PPU).
2822        match self.map_register(reg) {
2823            // $2000 / $2001 / $2003 / $2005 / $2006 are write-only; reads
2824            // return open-bus.
2825            0 | 1 | 3 | 5 | 6 => self.open_bus,
2826            2 => {
2827                // $2002 PPUSTATUS. High 3 bits are real; low 5 are open-bus.
2828                // `v` is the value as-of read-start (M2 high); it both feeds the
2829                // open-bus I/O-latch refresh below and is the base for the
2830                // CPU-visible return value. The Tier-1.1 read-end sample (below)
2831                // is applied ONLY to the returned byte, NOT to the open-bus
2832                // refresh — keeping the decay model byte-identical so real games
2833                // that read open bus after a `$2002` read are unaffected.
2834                // On a 2C05 the low 5 bits return the PPU identifier (copy
2835                // protection) instead of PPU open bus; on every other PPU they
2836                // are the open-bus latch (byte-identical to the legacy path).
2837                let low5 = if self.is_2c05 {
2838                    self.id_2c05 & 0x1F
2839                } else {
2840                    self.open_bus & 0x1F
2841                };
2842                let v = (self.status.bits() & 0xE0) | low5;
2843                // Clear VBL and the w toggle as a side effect.
2844                self.status.remove(PpuStatus::VBLANK);
2845                self.w = false;
2846                // R2 (master-clock R1 substrate): a $2002 read drops the /NMI
2847                // line UNCONDITIONALLY (Mesen2 `UpdateStatusFlag:588`
2848                // `ClearNmiFlag()`; TetaNES `read_status: nmi_pending=false`).
2849                // Correct under R1's on-time access (the read lands at the
2850                // access's exact dot); the CPU's φ2 edge detector sees the
2851                // level fall on the same access.
2852                {
2853                    self.nmi_line = false;
2854                }
2855                // Race: reading PPUSTATUS at exactly the cycle VBL would
2856                // have been set suppresses VBL + NMI for that frame. We
2857                // approximate the race window as scanline 241 dot 0 (the
2858                // dot before set) and dot 1 (the set dot).
2859                //
2860                // Session-18 / C1 attempt 16 (PPU-axis) investigated
2861                // tightening the predicate from `dot <= 1` to `dot == 0`
2862                // (matching Mesen2's `Core/NES/NesPpu.cpp::
2863                // UpdateStatusFlag()` `_cycle == 0` strict 1-dot window
2864                // and the nesdev wiki spec) but ROLLED IT BACK because
2865                // it did NOT flip the failing `cpu_interrupts_v2/{2,3,5}`
2866                // tests. The empirical oracle in
2867                // `ppu::tests::vbl_race_window_2002_read_sweep` (added
2868                // in Session-18) shows the predicate change cleanly
2869                // narrows the window AT the unit-test layer, but the
2870                // load-bearing axis at the integration-test layer is the
2871                // CPU-vs-PPU per-cycle access interleaving — a deeper
2872                // architectural surface that the Session-13 cold-boot
2873                // alignment closed at the FRAME-anchor level but NOT at
2874                // the INTRA-CYCLE phase level. See
2875                // `docs/audit/session-18-c1-attempt16-ppu-axis-rollback-2026-05-22.md`
2876                // and ADR-0002 §"Decision update (2026-05-22, Session-18)".
2877                // C1 attempt 18 (coordinated with the CPU-side φ1/φ2
2878                // split): when the access-reorder feature is enabled,
2879                // narrow the suppression window from `dot <= 1` to
2880                // `dot == 0` per Mesen2 line 590 + nesdev wiki spec.
2881                // The CPU-side shift puts our BIT $2002 reads at dot 1
2882                // (post-φ1-tick) instead of dot 0, and the dot-1 read
2883                // should NOT trigger suppression (it sees the
2884                // just-set VBL).  Both changes together close the
2885                // `cpu_interrupts_v2/{2,3,5}` sync_vbl divergence
2886                // documented in Session-17/18 audits.
2887                // R2 (mc-r1-substrate) narrows the race window to `dot == 0`
2888                // (Mesen2 `:590`) — on the on-time substrate a dot-1 read is a
2889                // normal post-set read; only dot-0 (one PPU clock before VBL
2890                // set) arms suppression.
2891                let in_race_window =
2892                    self.scanline == self.region.vblank_start_line() && self.dot == 0;
2893                if in_race_window {
2894                    self.suppress_vbl_this_frame = true;
2895                    // If NMI was already raised on dot 1 this same cycle,
2896                    // pull it back down too.
2897                    self.nmi_line = false;
2898                }
2899                // Reading PPUSTATUS only refreshes the upper 3 bits of the
2900                // open-bus latch (the bits sourced from the status register);
2901                // the lower 5 bits retain both their previous value AND their
2902                // decay timer.  See nesdev wiki "PPU registers" §"Open bus",
2903                // `cpu_dummy_writes_ppumem` test ROM (open_bus_read_test 2),
2904                // and `ppu_open_bus.nes` test 7
2905                // ("Reading $2002 shouldn't refresh low 5 bits of decay value").
2906                // Refresh groups 1 (bit 5) and 2 (bits 6-7) only.
2907                // Uses the read-start value `v` (the Tier-1.1 read-end mask is
2908                // applied to the return value only, below).
2909                self.refresh_open_bus(v, 0b110);
2910                // v2.0 Tier 1.1 — $2002 two-point intra-read flag sampling.
2911                // VBL (bit 7) is latched at read-start (M2 high) = the current
2912                // dot, already captured in `v`. The sprite-0 (bit 6) and
2913                // overflow (bit 5) flags are NOT latched; the CPU samples them
2914                // at read-end (M2 low), ~1.875 PPU dots later. A read straddling
2915                // the pre-render dot-1 flag-clear therefore returns VBL still set
2916                // while the sprite flags already read 0 — the AccuracyCoin
2917                // `$2002 flag timing` answer key `$E0,$E0,$80,$00`. Mask bits 6/5
2918                // on the returned byte when read-end lands at/after the clear
2919                // (TriCNES `EmulateUntilEndOfRead`). Returned-value-only so the
2920                // open-bus latch stays byte-identical for real games.
2921                if self.sprite_flags_clear_imminent(Self::STATUS_READ_END_DOTS) {
2922                    return v & !0x60;
2923                }
2924                v
2925            }
2926            4 => {
2927                // $2004 OAMDATA. Returns OAM[OAMADDR] without auto-increment.
2928                // Sprite attribute bytes (every 4th byte starting at offset 2)
2929                // have bits 2-4 unimplemented in OAM and always read as 0,
2930                // even though writes can store them.  See nesdev wiki "PPU
2931                // OAM" → "Byte 2 (attributes)".
2932                //
2933                // v2.0 Tier 1.2: while the screen is being drawn on a visible
2934                // scanline, $2004 returns the value the PPU is currently using
2935                // for sprite evaluation / loading (the OAM data bus), NOT
2936                // OAM[OAMADDR]. The isolated `ppu-oam-data-bus` model
2937                // (`oam_data_bus_read`) reproduces this per AccuracyCoin
2938                // `$2004 Stress`; see Mesen2 `NesPpu.cpp:298-313/361-380`.
2939                {
2940                    if self.oam_data_bus_is_live() {
2941                        let v = self.oam_data_bus_read();
2942                        self.touch_open_bus(v);
2943                        return v;
2944                    }
2945                }
2946                // v2.1.4 F2.3 — OAM-decay read hook (no-op at the default): a
2947                // non-rendering `$2004` read refreshes the row, or returns the
2948                // decayed pattern if it has gone stale. Must run before the read.
2949                self.oam_decay_on_read(self.oam_addr);
2950                let mut v = self.oam[self.oam_addr as usize];
2951                if (self.oam_addr & 0x03) == 0x02 {
2952                    v &= 0xE3;
2953                }
2954                // Per nesdev wiki "PPU registers" §$2004 + AccuracyCoin
2955                // "Address $2004 behavior" sub-tests 4 + 9: during dots
2956                // 1-64 of every rendered scanline (the secondary-OAM
2957                // clear phase) AND during dots 257-320 (the sprite-tile-
2958                // loading interval — also when the secondary-OAM bytes
2959                // are being read out to the shift registers), $2004
2960                // reads return $FF.
2961                //
2962                // (When `ppu-oam-data-bus` is on, the rendering case returns
2963                // above; this fallback covers the flag-off build + the
2964                // non-rendering paths.)
2965                if self.is_render_scanline()
2966                    && self.mask.rendering_enabled()
2967                    && ((1..=64).contains(&self.dot) || (257..=320).contains(&self.dot))
2968                {
2969                    v = 0xFF;
2970                }
2971                self.touch_open_bus(v);
2972                v
2973            }
2974            7 => {
2975                // Diagnostic: log where each $2007 read lands (scanline/dot/mask).
2976                if (4400..6200).contains(&self.frame) {
2977                    use core::sync::atomic::Ordering::Relaxed;
2978                    let i = read2007_diag::IDX.fetch_add(1, Relaxed) as usize;
2979                    if i < 1024 {
2980                        #[allow(clippy::cast_sign_loss)]
2981                        let sl = ((i32::from(self.scanline) + 1) as u32) & 0x1FF;
2982                        let packed = (sl << 18)
2983                            | ((u32::from(self.dot) & 0x1FFF) << 5)
2984                            | (u32::from(self.mask.rendering_enabled()) << 1)
2985                            | u32::from(self.is_render_scanline());
2986                        read2007_diag::LOG[i].store(packed, Relaxed);
2987                    }
2988                }
2989                // $2007 PPUDATA. Buffered for $0000-$3EFF; palette reads
2990                // bypass the buffer but still update it with the underlying
2991                // nametable mirror.
2992                let addr = self.v & 0x3FFF;
2993                let is_palette = addr >= 0x3F00;
2994                // W2: set when THIS read arms the PPUDATA SM countdown with the
2995                // defer-v-inc sub-knob on — the v-glitch increment then happens
2996                // at the TStep (the countdown landing dot), not here.
2997                let mut defer_v_inc = false;
2998                let result = if is_palette {
2999                    // Palette read: high 2 bits = open bus.
3000                    let palette = self.read_palette(addr);
3001                    let v_with_open_bus = (palette & 0x3F) | (self.open_bus & 0xC0);
3002                    // Buffer gets the underlying nametable byte (from CIRAM).
3003                    self.data_buffer = self.read_vram(bus, addr & 0x2FFF);
3004                    v_with_open_bus
3005                } else {
3006                    let r = self.data_buffer;
3007                    // During rendering the buffer is NOT loaded from a read at
3008                    // `v`. TriCNES (`Emulator.cs` `PPU_DATA_StateMachine` +
3009                    // the `$2007` CPU read): the read-END arms a latch cascade
3010                    // and the actual `PPU_ReadBuffer` reload happens ~4 dots
3011                    // later, latching the value the BG/sprite FETCH cadence
3012                    // drove on the VRAM bus at the LANDING dot (the fetch has
3013                    // bus priority). Modeled as a PPU-dot countdown consumed
3014                    // in `Ppu::tick`; the returned value stays the OLD buffer
3015                    // (the priming-read contract). Delay 0 = immediate latch
3016                    // of the current bus value.
3017                    if self.mask.rendering_enabled() && self.is_render_scanline() {
3018                        octal_trace::push(
3019                            octal_trace::K_R2007,
3020                            self.frame,
3021                            self.scanline,
3022                            self.dot,
3023                            u32::from(self.v & 0x3FFF),
3024                        );
3025                        let n = read2007_diag::RENDER_BUFFER_DOT_DELAY
3026                            .load(core::sync::atomic::Ordering::Relaxed)
3027                            as u8;
3028                        if n == 0 {
3029                            self.data_buffer = self.render_data_bus;
3030                        } else {
3031                            self.ppudata_sm_countdown = n;
3032                            // TriCNES `PPU_DATA_StateMachine_Half`: the TStep
3033                            // (v-glitch increment) fires at the SAME dot as the
3034                            // PD_RB buffer reload, AFTER it — so the fetches in
3035                            // the read-to-reload window still use the OLD `v`.
3036                            if read2007_diag::RENDER_BUFFER_DEFER_V_INC
3037                                .load(core::sync::atomic::Ordering::Relaxed)
3038                                != 0
3039                            {
3040                                self.ppudata_v_inc_pending = true;
3041                                defer_v_inc = true;
3042                            }
3043                        }
3044                    } else {
3045                        self.data_buffer = self.read_vram(bus, addr);
3046                    }
3047                    r
3048                };
3049                // Per nesdev "PPU rendering"
3050                // (https://www.nesdev.org/wiki/PPU_scrolling#$2007_reads_and_writes_during_rendering):
3051                // "Reading or writing PPUDATA during rendering (on the
3052                // pre-render line and the visible lines 0-239, only when
3053                // rendering is enabled) does not increment the address
3054                // normally, but instead increments both coarse X scroll
3055                // and Y scroll simultaneously, with normal wrapping."
3056                // This is the canonical "$2007 read w/ rendering" quirk
3057                // that AccuracyCoin's `PPU Behavior :: $2007 read w/
3058                // rendering` Test 2 brackets.
3059                //
3060                // W2 (`mc-ppu-2007-render-buffer` + defer-v-inc sub-knob): when
3061                // this read armed the PPUDATA SM countdown, the increment is
3062                // performed at the TStep (the countdown landing dot in
3063                // `Ppu::tick`) instead of here.
3064                let apply_inc_now = !defer_v_inc;
3065                if apply_inc_now {
3066                    if self.mask.rendering_enabled() && self.is_render_scanline() {
3067                        self.inc_hori_v();
3068                        self.inc_vert_v();
3069                    } else {
3070                        let inc = if self.ctrl.contains(PpuCtrl::VRAM_INCREMENT_32) {
3071                            32
3072                        } else {
3073                            1
3074                        };
3075                        self.v = self.v.wrapping_add(inc) & 0x7FFF;
3076                    }
3077                }
3078                // A12 transition can occur here.
3079                self.observe_a12(bus);
3080                if is_palette {
3081                    // Palette reads only refresh bits 0-5 of the decay model
3082                    // (palette is 6-bit); bits 6-7 retain their previous
3083                    // value AND timer.  Required by `ppu_open_bus.nes` test 9.
3084                    self.refresh_open_bus(result, 0b011);
3085                } else {
3086                    self.touch_open_bus(result);
3087                }
3088                result
3089            }
3090            _ => unreachable!(),
3091        }
3092    }
3093
3094    /// CPU register write.
3095    // Large by nature: an 8-way `$2000-$2007` register-dispatch match, each arm
3096    // carrying its own hardware-quirk handling (the v2.0.2 octal-latch `$2006`
3097    // hook nudged it past the 100-line lint threshold).
3098    #[allow(clippy::too_many_lines)]
3099    pub fn cpu_write_register<B: PpuBus>(&mut self, reg: u8, value: u8, bus: &mut B) {
3100        // Open-bus latch always picks up the written value.
3101        self.touch_open_bus(value);
3102        // 2C05 swaps $2000<->$2001 (no-op on every other PPU).
3103        match self.map_register(reg) {
3104            0 => {
3105                // $2000 PPUCTRL.
3106                if self.post_reset_mask_remaining > 0 {
3107                    return;
3108                }
3109                let prev_nmi_enable = self.ctrl.contains(PpuCtrl::NMI_ENABLE);
3110                self.ctrl = PpuCtrl::from_bits_truncate(value);
3111                // t bits 11-10 = nametable bits 1-0.
3112                self.t = (self.t & 0xF3FF) | ((u16::from(value) & 0x03) << 10);
3113                // NMI bit 0->1 transition while VBL set asserts NMI immediately.
3114                let new_nmi_enable = self.ctrl.contains(PpuCtrl::NMI_ENABLE);
3115                if !prev_nmi_enable && new_nmi_enable && self.status.contains(PpuStatus::VBLANK) {
3116                    self.nmi_line = true;
3117                }
3118                if !new_nmi_enable {
3119                    // Disabling NMI lowers the line.
3120                    self.nmi_line = false;
3121                }
3122            }
3123            1 => {
3124                // $2001 PPUMASK.
3125                if self.post_reset_mask_remaining > 0 {
3126                    return;
3127                }
3128                let was_rendering = self.mask.rendering_enabled();
3129                self.mask = PpuMask::from_bits_truncate(value);
3130                self.arm_oam_corruption_disable(was_rendering);
3131                // v2.0 Phase 6 (mc-ppu-subpos): arm the analog `$2001` BG-reload
3132                // delay. `self.mask` (and so the sprite-eval / shift / pixel
3133                // path) updates IMMEDIATELY — only the BG shift-register RELOAD
3134                // is gated on a value delayed `MASK_WRITE_DELAY` dots behind the
3135                // mask (TriCNES gates `PPU_Render_ShiftRegistersAndBitPlanes` ->
3136                // the reload on `PPU_Mask_Show*_Delayed`, while the per-half-dot
3137                // SHIFT runs on the IMMEDIATE mask). On a render re-enable edge
3138                // the shifter therefore advances for several dots (injecting the
3139                // serial-in '1') BEFORE the reload resumes — so one reload is
3140                // SKIPPED and the accumulated '1's reach the output (BG Serial
3141                // In) WITHOUT perturbing the sprite path (Stale Sprite Shift
3142                // Regs) or normal rendering (the reload value latched between
3143                // toggles equals the live mask -> byte-identical).
3144                // Freeze the BG-reload gate at its prior value for the analog
3145                // write-delay window; `tick` re-syncs it to the live mask once
3146                // the countdown settles.
3147                {
3148                    self.mask_write_delay =
3149                        MASK_WRITE_DELAY.load(core::sync::atomic::Ordering::Relaxed);
3150                }
3151            }
3152            2 => {
3153                // $2002 is read-only; writes only update the open-bus latch
3154                // (already done above) and otherwise have no effect.
3155            }
3156            3 => {
3157                // $2003 OAMADDR.
3158                self.oam_addr = value;
3159                // v2.1.7 P5 — OAMADDR ($2003) write-during-rendering OAM
3160                // corruption, modeled only on the earlier `Rp2c02G` revision
3161                // (default `Rp2c02H` skips this entirely → byte-identical). On
3162                // real "rev E+" 2C02 silicon, writing $2003 while rendering is
3163                // active corrupts one OAM "row"; RustyNES arms the shared
3164                // `CorruptOAM` row-copy (see `process_oam_corruption`) targeting
3165                // the row the write's high bits select, committed on the next
3166                // rendered dot. The `!oam_corruption_pending` guard defers to an
3167                // already-armed corruption (e.g. the rendering-disable model) so
3168                // the two sources never race. See `docs/ppu-2c02.md` (§OAMADDR
3169                // corruption) and `docs/accuracy-ledger.md` for the honesty note.
3170                if self.die_revision.models_oamaddr_corruption()
3171                    && self.mask.rendering_enabled()
3172                    && self.is_render_scanline()
3173                    && !self.oam_corruption_pending
3174                {
3175                    self.oam_corruption_pending = true;
3176                    self.oam_corruption_index = (value >> 3) & 0x1F;
3177                }
3178            }
3179            4 => {
3180                // $2004 OAMDATA write. Per nesdev §PPU OAM:
3181                //
3182                // - Outside rendering (or rendering disabled): write the
3183                //   value to OAM[OAMADDR] and increment OAMADDR by 1.
3184                // - During rendering (visible / pre-render scanline with
3185                //   rendering enabled): the write is BLOCKED (real chip
3186                //   does a glitchy "OAM read" instead, value discarded),
3187                //   but OAMADDR is still incremented by **4** (NOT 1) —
3188                //   the silicon's OAMADDR-bump-on-rendering-write quirk
3189                //   that AccuracyCoin's `Sprite Evaluation :: Misaligned
3190                //   OAM behavior` test (T-60-002, 2026-05-17) brackets.
3191                //
3192                // Pre-fix our impl always incremented by 1; matches the
3193                // outside-rendering path but is wrong during rendering.
3194                if self.mask.rendering_enabled() && self.is_render_scanline() {
3195                    // During-rendering quirk: OAMADDR += 4, then mask
3196                    // with $FC (clear bottom 2 bits — re-align to a
3197                    // 4-byte sprite boundary). Required for
3198                    // AccuracyCoin's "Address $2004 behavior" sub-test
3199                    // A which writes $2004 with OAMADDR=1 during
3200                    // rendering, then expects subsequent reads at
3201                    // OAMADDR=4 (= (1+4) & $FC) to read OAM[4].
3202                    self.oam_addr = self.oam_addr.wrapping_add(4) & 0xFC;
3203                } else {
3204                    self.oam[self.oam_addr as usize] = value;
3205                    // v2.3.2 "Lucid" — attribute only the branch that actually
3206                    // stores. The during-rendering branch above is BLOCKED by the
3207                    // hardware quirk, so recording it would attribute a byte to an
3208                    // instruction that demonstrably did not write it.
3209                    #[cfg(feature = "debug-hooks")]
3210                    if let Some(attrib) = self.write_attrib.as_mut() {
3211                        attrib.record_oam(self.oam_addr, self.attrib_pc, self.attrib_cycle, value);
3212                    }
3213                    // v2.1.4 F2.3 — OAM-decay write hook (no-op at the default):
3214                    // a direct `$2004` write refreshes the written row.
3215                    self.oam_decay_on_write(self.oam_addr);
3216                    self.oam_addr = self.oam_addr.wrapping_add(1);
3217                }
3218            }
3219            5 => {
3220                // $2005 PPUSCROLL.
3221                if self.post_reset_mask_remaining > 0 {
3222                    return;
3223                }
3224                if self.w {
3225                    // Second write — Y scroll.
3226                    self.t = (self.t & 0x8C1F)
3227                        | ((u16::from(value) & 0xF8) << 2)
3228                        | ((u16::from(value) & 0x07) << 12);
3229                    self.w = false;
3230                } else {
3231                    // First write — X scroll.
3232                    self.t = (self.t & 0xFFE0) | (u16::from(value) >> 3);
3233                    self.x = value & 0x07;
3234                    self.w = true;
3235                }
3236            }
3237            6 => {
3238                // $2006 PPUADDR.
3239                if self.post_reset_mask_remaining > 0 {
3240                    return;
3241                }
3242                if self.w {
3243                    // Second write — low byte; copy t to v.
3244                    self.t = (self.t & 0xFF00) | u16::from(value);
3245                    // v2.0.3 (ADR 0030, Option 1) — "Hybrid Addresses" the natural
3246                    // way. During rendering, a `$2006` second write does NOT copy
3247                    // `t -> v` immediately; it stages the delayed-`CopyV` countdown
3248                    // (`TriCNES` `PPU_Update2006Delay`). The `v = t` and the
3249                    // `address_bus = v` splice happen when the countdown lands
3250                    // (`Self::tick`), by which point the fetch cadence has advanced
3251                    // coarse-X and the per-group phase-0 nametable ALE has NATURALLY
3252                    // loaded `octal_latch` with the one-tile-ahead NT-low (`$19`),
3253                    // so the landing read splices `$2F00 | $19 = $2F19` with no
3254                    // reconstruction. Outside rendering the copy is immediate (the
3255                    // delay is unobservable there and would only risk shifting
3256                    // tightly-timed non-render code), so non-render behavior is
3257                    // unchanged.
3258                    let deferred_copy_v = self.mask.rendering_enabled()
3259                        && self.is_render_scanline()
3260                        // Only within the active BG-fetch window (visible dots
3261                        // 1..=256 + the dots-321..=336 prefetch). The "Hybrid
3262                        // Addresses" corruption can ONLY manifest when a background
3263                        // fetch is in flight to consume the stale octal latch; a
3264                        // `$2006` write during the sprite/HBlank interval
3265                        // (257..=320) has no BG-fetch consumer, so deferring `v = t`
3266                        // there would serve no accuracy purpose and only risk
3267                        // shifting the many commercial mid-frame scroll splits
3268                        // (SMB3's status-bar `$2006`/`$2005`, MMC5 titles) that
3269                        // write during HBlank. Narrowing to the fetch window keeps
3270                        // the delayed-`CopyV` surgical to the modeled artifact.
3271                        && ((1..=256).contains(&self.dot) || (321..=336).contains(&self.dot))
3272                        && {
3273                            // Alignment-dependent delay (TriCNES uses 4 for three of
3274                            // four CPU/PPU phases, 5 for one). The `$2006` write is
3275                            // applied at the START of a CPU cycle in RustyNES's
3276                            // lockstep bus (before that cycle's 3 PPU ticks); the
3277                            // corrupted NT read is the phase-1 dot of the fetch group
3278                            // one coarse-X past the write. Empirically calibrated
3279                            // against the TriCNES per-dot trace (see the campaign
3280                            // plan); the AccuracyCoin test also retries across
3281                            // frames/alignments so at least one alignment lands.
3282                            self.copy_v_delay = COPY_V_DELAY;
3283                            octal_trace::push(
3284                                octal_trace::K_W2006,
3285                                self.frame,
3286                                self.scanline,
3287                                self.dot,
3288                                u32::from(self.t & 0x3FFF),
3289                            );
3290                            true
3291                        };
3292                    if !deferred_copy_v {
3293                        self.v = self.t;
3294                        // PPUADDR write can flip A12.
3295                        self.observe_a12(bus);
3296                    }
3297                    self.w = false;
3298                } else {
3299                    // First write — high byte (clears bit 14 of t).
3300                    self.t = (self.t & 0x00FF) | ((u16::from(value) & 0x3F) << 8);
3301                    self.w = true;
3302                }
3303            }
3304            7 => {
3305                // $2007 PPUDATA write. Same rendering quirk as the
3306                // read path (see `cpu_read_register` case 7 docstring):
3307                // writes during rendering increment both coarse-X and
3308                // Y scroll instead of the normal `inc` value.
3309                let addr = self.v & 0x3FFF;
3310                if addr >= 0x3F00 {
3311                    self.write_palette(addr, value);
3312                } else {
3313                    self.write_vram(bus, addr, value);
3314                }
3315                if self.mask.rendering_enabled() && self.is_render_scanline() {
3316                    self.inc_hori_v();
3317                    self.inc_vert_v();
3318                } else {
3319                    let inc = if self.ctrl.contains(PpuCtrl::VRAM_INCREMENT_32) {
3320                        32
3321                    } else {
3322                        1
3323                    };
3324                    self.v = self.v.wrapping_add(inc) & 0x7FFF;
3325                }
3326                self.observe_a12(bus);
3327            }
3328            _ => unreachable!(),
3329        }
3330    }
3331
3332    /// Address-bus A12 = `v` bit 12 during `$0000-$3FFF` accesses. Notify the
3333    /// mapper on every transition.  Also called by `observe_a12_addr` for
3334    /// the actual pattern fetch addresses (background and sprite fetches
3335    /// directly read CHR via the address bus, not via `v`).
3336    fn observe_a12<B: PpuBus>(&mut self, bus: &mut B) {
3337        let level = (self.v & 0x1000) != 0;
3338        if level != self.last_a12_level {
3339            bus.notify_a12(level);
3340            self.last_a12_level = level;
3341        }
3342    }
3343
3344    /// Report a background fetch group's A12 level where the `NESdev` MMC3
3345    /// page puts it: the pattern fetches' high level at group phase 3 (dot
3346    /// 324 for the next line's first prefetched tile, dot 4 for a line's own
3347    /// first group), one dot before the pattern-low fetch's ALE dot, and the
3348    /// return to the nametable fetch's low level at phase 7.
3349    ///
3350    /// The page: "if the BG uses `$1000`, and the sprites use `$0000`, the
3351    /// IRQ counter should decrement on PPU cycle 324 of the previous
3352    /// scanline", and for the opposite arrangement "on PPU cycle 260". The
3353    /// sprite path already reported at that convention (its rises land at
3354    /// 260, 268, ... 316). The background reported at its READ dots (phase
3355    /// 5 for the pattern, phase 1 for the nametable), two dots later than
3356    /// the page, so a rise that landed on the first dot of a CPU cycle's
3357    /// catch-up was seen by an MMC3 a cycle late. blargg `4-scanline_timing`
3358    /// failed at sub-test 9 ("Scanline 0 IRQ should occur sooner when
3359    /// `$2000=$10`") on exactly that case; with this it fails at 12, the
3360    /// sub-test the `MiSTer` DUT fails (T-MMC3-BG-A12).
3361    ///
3362    /// The read halves still report their own addresses, and with the level
3363    /// already set those reports change nothing. Only the level the mapper
3364    /// sees, and when, moves: with the background at `$0000` (almost every
3365    /// MMC3 game) both reports here are low and nothing changes.
3366    fn observe_bg_a12_lead<B: PpuBus>(&mut self, bus: &mut B, phase: u16) {
3367        match phase {
3368            3 => {
3369                let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
3370                self.observe_a12_addr(bus, bg_table);
3371            }
3372            7 => self.observe_a12_addr(bus, 0x2000),
3373            _ => {}
3374        }
3375    }
3376
3377    /// Notify the mapper of an A12 transition implied by an explicit
3378    /// pattern-table fetch address (BG / sprite fetches that bypass `v`).
3379    fn observe_a12_addr<B: PpuBus>(&mut self, bus: &mut B, addr: u16) {
3380        let level = (addr & 0x1000) != 0;
3381        if level != self.last_a12_level {
3382            bus.notify_a12(level);
3383            self.last_a12_level = level;
3384        }
3385    }
3386
3387    /// Read from PPU memory `$0000-$3EFF` honoring CIRAM ownership: CHR
3388    /// (`$0000-$1FFF`) goes to the bus/mapper; nametable (`$2000-$3EFF`)
3389    /// reads come from the PPU-owned CIRAM through the mapper-supplied
3390    /// mirroring map.
3391    ///
3392    /// The bus is consulted via `peek_nametable` first; mappers like MMC5
3393    /// in fill mode or ExRAM-as-nametable mode synthesize the byte
3394    /// directly. Only when the bus declines (`None`) do we hit CIRAM.
3395    // `&mut self` is required under `mc-ppu-2007-render-buffer` (it latches
3396    // `self.render_data_bus` below); clippy's needless-pass-by-ref-mut only fires
3397    // on the default build where that cfg is off, so allow it here.
3398    fn read_vram<B: PpuBus>(&mut self, bus: &mut B, addr: u16) -> u8 {
3399        // v2.9.7 — every read drives its address onto the PPU bus, so every
3400        // read reports its A12 level. Before v2.9.7 only pattern fetches and
3401        // `v` did, so the garbage nametable reads of the sprite window never
3402        // pulled A12 low, and the eight A12 pulses of each rendered line
3403        // reached the mapper as one. MMC3's filter ignores the short lows, so
3404        // it never showed there. Boards that count raw edges were starved
3405        // eightfold: Acclaim's MC-ACC (a divide-by-8 on every edge) and
3406        // mapper 91 submapper 0 ("64 unfiltered rises"). Pinned by
3407        // `a12_reports_the_hardware_stream_and_an_mmc3_filter_still_sees_241`.
3408        // `observe_a12_addr` calls the bus only on a change of level.
3409        self.observe_a12_addr(bus, addr & 0x3FFF);
3410        self.read_vram_unobserved(bus, addr)
3411    }
3412
3413    /// [`Self::read_vram`] without the A12 report: the read itself, for the
3414    /// one caller whose A12 edge would land in the wrong order (the sprite
3415    /// window's second garbage nametable read; see `tick_sprite_fetch_read`).
3416    #[allow(clippy::needless_pass_by_ref_mut)]
3417    fn read_vram_unobserved<B: PpuBus>(&mut self, bus: &mut B, addr: u16) -> u8 {
3418        let a = addr & 0x3FFF;
3419        // Every PPU bus read passes through here -- pattern fetches, nametable
3420        // and attribute fetches, sprite pattern fetches, and `$2007`. Capturing
3421        // at the choke point rather than at each call site is what makes the
3422        // trace complete by construction: a fetch added later cannot forget to
3423        // record itself.
3424        #[cfg(feature = "ppu-fetch-trace")]
3425        if let Some(trace) = self.fetch_trace.as_mut() {
3426            trace.push(crate::fetch_trace::FetchRecord {
3427                frame: u32::try_from(self.frame).unwrap_or(u32::MAX),
3428                scanline: self.scanline,
3429                dot: self.dot,
3430                addr: a,
3431            });
3432        }
3433        let val = if a < 0x2000 {
3434            bus.ppu_read(a)
3435        } else {
3436            // Mirror $3000-$3EFF to $2000-$2EFF, unless the cartridge has RAM
3437            // there (v2.9.6: GTROM, UNROM 512 four-screen). Asked only for a
3438            // `$3xxx` address, which rendering never produces.
3439            let nt_addr = if a >= 0x3000 && !bus.nametable_unfolded() {
3440                a - 0x1000
3441            } else {
3442                a
3443            };
3444            if let Some(v) = bus.peek_nametable(nt_addr) {
3445                v
3446            } else {
3447                let off = bus.nametable_address(nt_addr) as usize;
3448                self.ciram[off & 0x07FF]
3449            }
3450        };
3451        // The VRAM data bus latches every read (the rendering fetches drive it);
3452        // a `$2007` read during rendering returns this, not a read at `v`.
3453        {
3454            self.render_data_bus = val;
3455        }
3456        // v2.0.3 (ADR 0030) — TriCNES `FetchPPU`: after the read the multiplexed
3457        // bus's low 8 bits hold the DATA (AD7-0). Crucially, `octal_latch` is NOT
3458        // refreshed here — it retains the ADDRESS low it was loaded with at ALE, so
3459        // a following `$2007`-read ALE-overlap freezes it on this stale data byte
3460        // (the "ALE + Read" corruption; the latch is managed by `ale_splice` /
3461        // `drive_bus`).
3462        val
3463    }
3464
3465    /// W2 ($2007 Stress) — per-dot sprite-tile fetch read cadence (dots
3466    /// 257-320), feeding `render_data_bus` for the deferred `$2007` `PPUDATA`
3467    /// buffer reload. Per `AccuracyCoin` `$2007 Stress` (and `TriCNES`'s per-dot
3468    /// PPU), each 8-dot sprite slot does TWO nametable reads in a row (not
3469    /// NT+AT) then the sprite PT-lo / PT-hi. Both garbage NT reads use the
3470    /// (horizontally-reset) `v` address — the sprite-fetch interval does no
3471    /// coarse-X increment, so it is constant across all 8 slots. Reads land
3472    /// on the slot-local odd dots (1,3,5,7). The PT bytes come from the raw
3473    /// stash captured by `fetch_sprite_tile` (no fresh CHR read, so no new
3474    /// A12/mapper events); the NT reads go through `read_vram`, which latches
3475    /// `render_data_bus` itself.
3476    fn tick_sprite_fetch_read<B: PpuBus>(&mut self, bus: &mut B) {
3477        let local = (self.dot - 257) % 8;
3478        let slot = ((self.dot - 257) / 8) as usize;
3479        match local {
3480            1 | 3 => {
3481                // Slot 0's FIRST garbage NT read straddles the dot-257
3482                // copy-hori boundary: its ALE (dot 257) latched the OLD `v`
3483                // address (`ppudata_spr0_nt_addr`). Every later garbage NT
3484                // read uses the (horizontally reset) live `v`.
3485                let nt = if local == 1 && slot == 0 {
3486                    self.ppudata_spr0_nt_addr
3487                } else {
3488                    0x2000 | (self.v & 0x0FFF)
3489                };
3490                // `read_vram` latches `render_data_bus` with the value read.
3491                //
3492                // A12 (v2.9.7): on hardware the slot's two garbage reads hold
3493                // A12 low for dots 257-260 of the slot, and the pattern fetch
3494                // raises it at 261. Here the pattern fetch is collapsed into
3495                // `fetch_sprite_tile` at the slot's dot 260, the same dot as the
3496                // second garbage read (local 3), and runs BEFORE it. Reporting
3497                // that read's A12 would lower A12 again after the rise and leave
3498                // it low for eight dots instead of high, which lets an MMC3
3499                // filter count a second rise per slot (mmc3_test 2 "clocked 241
3500                // times" fails). So the first read (local 1) reports the fall,
3501                // the pattern fetch the rise, and the second read stays silent:
3502                // the same edges, one per slot, as the hardware.
3503                let _ = if local == 1 {
3504                    self.read_vram(bus, nt)
3505                } else {
3506                    self.read_vram_unobserved(bus, nt)
3507                };
3508            }
3509            5 if slot < 8 => self.render_data_bus = self.spr_fetch_lo_raw[slot],
3510            7 if slot < 8 => self.render_data_bus = self.spr_fetch_hi_raw[slot],
3511            _ => {}
3512        }
3513    }
3514
3515    /// Write to PPU memory `$0000-$3EFF`. Mirrors [`Self::read_vram`].
3516    fn write_vram<B: PpuBus>(&mut self, bus: &mut B, addr: u16, value: u8) {
3517        let a = addr & 0x3FFF;
3518        if a < 0x2000 {
3519            bus.ppu_write(a, value);
3520        } else {
3521            let nt_addr = if a >= 0x3000 && !bus.nametable_unfolded() {
3522                a - 0x1000
3523            } else {
3524                a
3525            };
3526            // Give the mapper a chance to absorb the write (ExRAM
3527            // nametables, fill-mode drops, etc.). If declined, write CIRAM.
3528            if !bus.write_nametable(nt_addr, value) {
3529                let off = bus.nametable_address(nt_addr) as usize;
3530                self.ciram[off & 0x07FF] = value;
3531                // v2.3.2 "Lucid" — attribute the byte to the instruction that
3532                // stored it. Recorded here rather than at the CPU-write boundary
3533                // because only this site knows the resolved physical offset: the
3534                // caller wrote `$2007`, and `v` plus the mapper's mirroring is
3535                // what turned that into `off`.
3536                #[cfg(feature = "debug-hooks")]
3537                if let Some(attrib) = self.write_attrib.as_mut() {
3538                    attrib.record_ciram(off, self.attrib_pc, self.attrib_cycle, value);
3539                }
3540            }
3541        }
3542    }
3543
3544    /// Read palette RAM. Mirrors:
3545    ///   $3F10/$14/$18/$1C → $3F00/$04/$08/$0C
3546    ///   anything past $3F1F mirrors back into the 32-byte window.
3547    const fn read_palette(&self, addr: u16) -> u8 {
3548        let idx = palette_index(addr);
3549        // Apply the greyscale mask if PPUMASK bit 0 is set.
3550        let raw = self.palette_ram[idx];
3551        if self.mask.contains(PpuMask::GREYSCALE) {
3552            raw & 0x30
3553        } else {
3554            raw
3555        }
3556    }
3557
3558    // Const-promotable only when `debug-hooks` is off (the attribution branch
3559    // below dereferences a `Box`, which is not const). Allowing the lint keeps
3560    // ONE definition instead of two cfg'd copies of the same three lines.
3561    #[allow(clippy::missing_const_for_fn)]
3562    fn write_palette(&mut self, addr: u16, value: u8) {
3563        let idx = palette_index(addr);
3564        // Palette is 6-bit storage.
3565        self.palette_ram[idx] = value & 0x3F;
3566        // v2.3.2 "Lucid" — record the MASKED value, so the attribution matches
3567        // what a later read returns rather than what the CPU put on the bus.
3568        // `idx` is post-mirroring, so an attribution looked up through `$3F10`
3569        // and through `$3F00` resolves to the same record — correct, since they
3570        // are the same byte.
3571        #[cfg(feature = "debug-hooks")]
3572        if let Some(attrib) = self.write_attrib.as_mut() {
3573            attrib.record_palette(idx, self.attrib_pc, self.attrib_cycle, value & 0x3F);
3574        }
3575    }
3576
3577    /// Re-point the rendering gate to the swept depth, and hand back the
3578    /// 1-dot-delayed value as it stood when the dot began.
3579    ///
3580    /// Called at the TOP of [`Self::tick`], before `rendering_gate` and every
3581    /// other consumer reads `rendering_enabled_delayed` this dot. Depth 1 is
3582    /// the shipped behaviour and leaves the field exactly as the tick-end
3583    /// assignment left it, so the default path is untouched by construction
3584    /// rather than by claim.
3585    ///
3586    /// The return value is the pipeline's stage N-1 and MUST be the value
3587    /// handed to [`Self::render_gate_end_dot`]: at depth >= 2 this function
3588    /// overwrites the field with stage N-2, so reading the field back at
3589    /// tick-end assigns `render_gate_prev2` to itself and the pipeline freezes
3590    /// at its power-on value instead of shifting. That is the defect found in
3591    /// review on #506 and the reason the shift is a named pair rather than two
3592    /// bare assignments a hundred lines apart.
3593    #[cfg(feature = "phi2-write-sweep")]
3594    const fn render_gate_begin_dot(&mut self, depth: u8) -> bool {
3595        let prev1 = self.rendering_enabled_delayed;
3596        match depth {
3597            0 => self.rendering_enabled_delayed = self.mask.rendering_enabled(),
3598            1 => {}
3599            _ => self.rendering_enabled_delayed = self.render_gate_prev2,
3600        }
3601        prev1
3602    }
3603
3604    /// Shift stage N-1 into stage N-2. Called at the END of [`Self::tick`],
3605    /// beside the `rendering_enabled_delayed = rendering` that fills stage N-1,
3606    /// with the value [`Self::render_gate_begin_dot`] returned for this dot.
3607    #[cfg(feature = "phi2-write-sweep")]
3608    const fn render_gate_end_dot(&mut self, prev1: bool) {
3609        self.render_gate_prev2 = prev1;
3610    }
3611
3612    /// Tick exactly one dot.
3613    #[allow(clippy::too_many_lines)] // the per-dot FSM + the ppu-oam-data-bus tick hook
3614    #[allow(clippy::cognitive_complexity)] // + the ppu-sprite-shifter-counter render-toggle branches
3615    pub fn tick<B: PpuBus>(&mut self, bus: &mut B) {
3616        // Advance the dot/scanline FSM first, then handle per-dot events at
3617        // the post-advance position.
3618        self.advance_dot();
3619
3620        // === v2.1.8 A1 — specialized visible-scanline fast dot path ===
3621        //
3622        // The per-dot `tick` FSM below is the emulator's single hottest
3623        // function (`Ppu::tick` ~46% of a representative frame's self-time,
3624        // `docs/performance.md`). The overwhelming majority of its 89,342
3625        // per-frame invocations are visible-scanline BG-render dots whose
3626        // surrounding event/bookkeeping branches are all statically dead —
3627        // no scanline-241 VBL set, no pre-render clear, no OAM-corruption
3628        // edge, no PPUDATA state machine in flight, no `$2006` copy-V or
3629        // PPUMASK write delay pending, rendering stably enabled. This gate
3630        // detects that regime cheaply and, when the (default-OFF) runtime
3631        // knob is on, dispatches to [`Self::tick_visible_render_fast`], which
3632        // runs the *identical* helper sequence with the dead branches pruned.
3633        //
3634        // BYTE-IDENTITY: the fast handler is byte-identical BY CONSTRUCTION —
3635        // it calls the same helpers (`tick_oam_corruption`,
3636        // `tick_sprite_eval_per_dot`, `tick_oam_bus`, `reload_bg_shift_regs`,
3637        // the `ale_drive_*` / `fetch_*` pair, `inc_hori_v`, `inc_vert_v`,
3638        // `emit_pixel`, `shift_bg`) in the same order the general path would
3639        // for a dot satisfying the guard, and executes NONE of the branches
3640        // the guard proves un-taken. The guard is conservative: any doubt
3641        // (delay counters non-zero, corruption armed, cache cold, rendering
3642        // toggling) falls through to the exact path below. Empirically pinned
3643        // bit-for-bit by the differential test
3644        // (`crates/rustynes-test-harness/tests/fast_dotloop_diff.rs`) and the
3645        // full AccuracyCoin / visual-regression / nestest oracle.
3646        //
3647        // Compiled out under `ppu-state-trace` (whose end-of-tick hook must
3648        // observe every dot); under that feature the knob is inert.
3649        // Guard-condition ORDER is tuned to fail fast (short-circuit) for the
3650        // common non-covered dots so the flag costs little when it does not
3651        // apply: the dot-range and rendering-enabled tests eliminate the
3652        // out-of-window and rendering-disabled dots first (a rendering-disabled
3653        // frame — e.g. the `flowing_palette` all-64-colour backdrop-override
3654        // demo — bails at `rendering_enabled()` before the more numerous
3655        // sub-dot-disturbance checks). AND is commutative, so the reorder is
3656        // byte-identity-neutral. NOTE: `cached_visible` is only meaningful once
3657        // the classification cache is warm, so `scanline == flags_cached_scanline`
3658        // MUST be tested before it.
3659        #[cfg(not(feature = "ppu-state-trace"))]
3660        if self.fast_dotloop
3661            && self.dot <= 256
3662            && self.dot >= 1
3663            // Rendering stably enabled: immediate == 1-dot-delayed == previous
3664            // dot's value, so `rendering`, `rendering_gate`, `bg_reload_render`
3665            // and the shift gate all collapse to `true` with no edge to model.
3666            && self.mask.rendering_enabled()
3667            && self.rendering_enabled_delayed
3668            // v2.6.18: the fast body performs its own dot-256 `inc_vert_v`,
3669            // which reads a TWO-dot rendering history, while every other
3670            // rendering term here proves only ONE -- `rendering_enabled_delayed`
3671            // and `prev_rendering_enabled` are both "rendering as of the
3672            // previous dot", assigned together. A mask that turned on at dot
3673            // 255 satisfies all of them at dot 256 with the two-dot view still
3674            // `false`, and the fast body would then increment where the general
3675            // path does not.
3676            //
3677            // HONESTY NOTE, so the next mutation pass does not re-investigate:
3678            // deleting this term is **NOT CAUGHT** by anything in this tree --
3679            // not `fast_dotloop_is_byte_identical_across_corpus`, not the
3680            // AccuracyCoin battery, and not
3681            // `fast_dotloop_is_byte_identical_when_an_enable_lands_beside_dot_256`,
3682            // which was written for exactly this and sweeps twelve power-on
3683            // alignments of the one ROM that writes `$2001` next to dot 256.
3684            // Writes land on CPU-cycle boundaries, so the reachable effect dots
3685            // are three apart, and no stimulus here lands one at 254->255.
3686            //
3687            // It is kept anyway, and that is not stubbornness: the term can only
3688            // make the fast path be taken LESS often, and the general path is
3689            // the reference, so its presence cannot cause a divergence while its
3690            // absence is a latent one waiting for a ROM that writes there.
3691            && self.rendering_enabled_delayed2
3692            && self.prev_rendering_enabled
3693            // Scanline classification cache warm (dot 0 of the line, taken on
3694            // the general path, warms it) AND this is a visible scanline.
3695            && self.scanline == self.flags_cached_scanline
3696            && self.cached_visible
3697            // No sub-dot disturbance in flight.
3698            && self.copy_v_delay == 0
3699            && self.mask_write_delay == 0
3700            && self.ppudata_sm_countdown == 0
3701            && !self.oam_corruption_pending
3702            && !self.oam_corruption_disabled
3703            && !self.oam_corruption_disabled_instant
3704            // Depth-2 sweep only; a no-op constant in the shipped build.
3705            && fast_dot_paths_valid()
3706        {
3707            #[cfg(feature = "ppu-fetch-trace")]
3708            {
3709                self.fast_path_hits = self.fast_path_hits.saturating_add(1);
3710            }
3711            self.tick_visible_render_fast(bus);
3712            return;
3713        }
3714
3715        // === v2.2.3 P2 — specialized IDLE-LINE fast dot path ===
3716        //
3717        // A1 (above) covers visible dots 1..=256 — 61,440 of the 89,342 NTSC
3718        // dots, 68.8%. The remaining 31.2% still walk the whole general body.
3719        // The cheapest slice of that remainder to prove is the **idle line**:
3720        // the post-render line (240) plus every vblank line except the VBL-set
3721        // line 241, i.e. 20 of 262 lines / 6,820 dots per frame.
3722        //
3723        // On such a dot the general path below reduces, provably, to exactly
3724        // three assignments — every other branch is gated on `render_line`,
3725        // `visible`, `pre_render`, `scanline == vblank_start_line()`, or a
3726        // disturbance counter this guard requires to be zero:
3727        //
3728        //   * `bg_reload_render = mask.rendering_enabled()` (the
3729        //     `mask_write_delay == 0` arm),
3730        //   * `prev_rendering_enabled = rendering`,
3731        //   * `rendering_enabled_delayed = rendering`.
3732        //
3733        // `tick_idle_line_fast` performs precisely those, in that order, from
3734        // the same single `mask.rendering_enabled()` read — so it is
3735        // byte-identical BY CONSTRUCTION, on the same terms as A1.
3736        //
3737        // The guard requires the classification cache to be WARM for this
3738        // scanline, which is what makes `cached_idle_line` trustworthy: dot 0
3739        // of every line misses the cache and takes the general path (warming
3740        // it), so the fast path serves dots 1..=340 — 340 of each idle line's
3741        // 341 dots.
3742        //
3743        // It also requires the three sub-dot disturbance countdowns to be
3744        // idle. `$2006` (`copy_v_delay`) and `$2001` (`mask_write_delay`) are
3745        // load-bearing: both are perfectly legal during vblank — that is when
3746        // most games issue them — and each has real work to do on landing.
3747        // `ppudata_sm_countdown` is BELT-AND-BRACES: it is armed only under
3748        // `mask.rendering_enabled() && is_render_scanline()` (see the `$2007`
3749        // read handler), so it cannot currently be live on an idle line at all.
3750        // It is tested anyway so the guard's correctness is self-evident from
3751        // the guard itself, rather than resting on an invariant enforced three
3752        // hundred lines away that a future change could quietly break. One
3753        // comparison is a fair price for that.
3754        //
3755        // NOTE for anyone extending this: the three assignments in
3756        // `tick_idle_line_fast` are, given this guard, provably redundant —
3757        // the mask cannot change without a `$2001` write, which arms
3758        // `mask_write_delay` and routes the affected dots through the general
3759        // path, so the values are already correct on every dot the fast path
3760        // serves. Verified empirically: deleting any one of them leaves the
3761        // whole differential suite green. They are kept regardless, because
3762        // "runs the same assignments in the same order" is a claim that can be
3763        // checked by reading twenty lines, whereas "these stores are dead" is a
3764        // reachability argument that must be re-derived every time the guard
3765        // moves. Three stores per idle dot is not worth trading that away.
3766        //
3767        // Compiled out under `ppu-state-trace`, whose end-of-tick hook must
3768        // observe every dot (same treatment as A1).
3769        #[cfg(all(feature = "ppu-idle-line-fast", not(feature = "ppu-state-trace")))]
3770        if self.fast_dotloop
3771            && self.scanline == self.flags_cached_scanline
3772            && self.cached_idle_line
3773            && self.copy_v_delay == 0
3774            && self.mask_write_delay == 0
3775            && self.ppudata_sm_countdown == 0
3776            // Depth-2 sweep only; a no-op constant in the shipped build.
3777            && fast_dot_paths_valid()
3778        {
3779            self.tick_idle_line_fast();
3780            return;
3781        }
3782
3783        // v2.0.3 (ADR 0030, Option 1) — the delayed-`CopyV` landing (`TriCNES`
3784        // `Emulator.cs:1684-1704`). Ticked at the TOP of the dot, BEFORE the fetch
3785        // dispatch, so the `address_bus = v` splice is in place for THIS dot's
3786        // nametable read (the corrupted "Hybrid Addresses" fetch). The phase-0 NT
3787        // ALE of the corrupt group ran on the PREVIOUS dot and already loaded
3788        // `octal_latch` with the one-tile-ahead NT-low, so the armed splice at the
3789        // read below yields `(v & 0x3F00) | octal_latch = $2F19` naturally.
3790        if self.copy_v_delay > 0 {
3791            self.copy_v_delay -= 1;
3792            if self.copy_v_delay == 0 {
3793                self.v = self.t;
3794                self.address_bus = self.v;
3795                // Preserve the `$2006`-write A12 edge (MMC3 timing). It is delayed
3796                // by the countdown vs the flag-off immediate copy, but `$2006`
3797                // writes during active render are rare and A12 during render is
3798                // dominated by the dot-260 sprite fetch, so this stays inside the
3799                // fetch-address-derived-timing budget (verified by the battery).
3800                self.observe_a12(bus);
3801            }
3802        }
3803
3804        // v2.0 Phase 6 (mc-ppu-subpos): track the BG-reload gate. It follows the
3805        // live `self.mask` rendering bit EXCEPT during the analog `$2001`
3806        // write-delay window, where it stays frozen at the prior value (TriCNES
3807        // `PPU_Update2001Delay` -> `PPU_Mask_Show*_Delayed`). Re-syncing to the
3808        // live mask when settled keeps it consistent under a direct mask set
3809        // (unit tests / save-state restore). Done at the top of the dot, before
3810        // the reload runs; the live `self.mask` is unaffected.
3811        if self.mask_write_delay > 0 {
3812            self.mask_write_delay -= 1;
3813        } else {
3814            self.bg_reload_render = self.mask.rendering_enabled();
3815        }
3816
3817        // v1.4.0 Workstream F (F1): the scanline-classification flags are pure
3818        // functions of `self.scanline` + `self.region`, so recompute them only
3819        // when the scanline changes; every other dot reads the cached copies.
3820        // Byte-identical (same values), self-healing on reset / restore.
3821        if self.scanline != self.flags_cached_scanline {
3822            self.cached_visible =
3823                self.scanline >= 0 && self.scanline <= self.region.last_visible_line();
3824            self.cached_pre_render = self.scanline == self.region.prerender_line();
3825            self.cached_render_line = self.cached_visible || self.cached_pre_render;
3826            // v2.2.3 P2 — an "idle" line: neither visible nor pre-render, and not
3827            // the VBL-set line. That is the post-render line (240) plus every
3828            // vblank line except 241 — 20 of the 262 NTSC lines. On such a line
3829            // no dot fetches, renders, evaluates sprites, or raises an event, so
3830            // the whole per-dot body collapses to the rendering-flag bookkeeping
3831            // (see `tick_idle_line_fast`). Cached here with the other
3832            // classification flags because it is the same pure function of
3833            // `scanline` + `region` and shares their `flags_cached_scanline` key.
3834            #[cfg(feature = "ppu-idle-line-fast")]
3835            {
3836                self.cached_idle_line =
3837                    !self.cached_render_line && self.scanline != self.region.vblank_start_line();
3838            }
3839            self.flags_cached_scanline = self.scanline;
3840        }
3841        let visible = self.cached_visible;
3842        let pre_render = self.cached_pre_render;
3843        let render_line = self.cached_render_line;
3844        let rendering = self.mask.rendering_enabled();
3845        // v2.6.18 derivation: re-point the delayed gate to the swept depth
3846        // BEFORE `rendering_gate` and every other consumer reads it this dot.
3847        // Depth 1 leaves it exactly as the tick-end assignment left it, so the
3848        // default path is untouched BY CONSTRUCTION rather than by claim.
3849        // The 1-dot-delayed value AS THIS DOT STARTED, captured before the
3850        // re-point below overwrites the field. The tick-end shift needs stage
3851        // N-1 to feed stage N-2, and reading the field back there after a
3852        // depth-2 re-point assigns `render_gate_prev2` to ITSELF — freezing the
3853        // pipeline at its power-on `false` for the whole run instead of
3854        // shifting. Found in review on #506, after the depth sweep had already
3855        // been run: every `lag >= 2` cell measured a permanently-disabled gate
3856        // rather than a two-dot delay. `render_gate_lag_shifts_a_two_dot_pipeline`
3857        // fails without this line.
3858        #[cfg(feature = "phi2-write-sweep")]
3859        let render_gate_prev1 =
3860            self.render_gate_begin_dot(RENDER_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed));
3861        // v2.0 (ae30785): the fetch/shift/sprite-eval pipeline gates on the
3862        // 1-PPU-dot-delayed rendering value under `ppu-sprite-shifter-counter`
3863        // (a mid-scanline `$2001` toggle takes effect one dot later — Stale
3864        // BG/Sprite). Default build = the immediate value (byte-identical).
3865        let rendering_gate = self.rendering_enabled_delayed;
3866        // Stage 2, captured HERE for the same reason stage 1 is: the pipeline
3867        // update below runs BEFORE the dot-event section, so a consumer that
3868        // reads the field after it sees this dot's value rather than the
3869        // delayed one. Reading `self.rendering_enabled_delayed2` at the dot-256
3870        // site gave rendering ONE dot ago and the increment kept firing.
3871        let rendering_gate2 = self.rendering_enabled_delayed2;
3872
3873        // OAM corruption (TriCNES eval-pointer model). The disable edge
3874        // itself is armed by the `$2001` write (see the PPUMASK handler);
3875        // the index is captured against the live secondary-OAM eval
3876        // pointer during the dots 1-64 window (`capture_oam_corruption`,
3877        // called from the sprite-eval FSM); and the actual corruption is
3878        // committed when rendering RE-ENABLES on a render line, or at the
3879        // pre-render line. Here we only retain the unrelated BG-shifter
3880        // fix-up that the prior model happened to share the 1->0 edge with.
3881        if render_line && rendering != self.prev_rendering_enabled && !rendering {
3882            // v2.0 (ppu-sprite-shifter-counter): if rendering is disabled
3883            // mid-pre-fetch (dots 329-336, the SECOND fetch group after the
3884            // dot-329 reload), the in-progress group's pending `<<= 8` would be
3885            // skipped by the now-gated pipeline, freezing the just-reloaded tile
3886            // in the BG shifter's bits 0-7. Complete it ONCE here (one-time on
3887            // the 1->0 edge) so the tile lands in bits 8-15 and surfaces at the
3888            // correct pixel on re-enable (Stale Sprite Shift Regs t5/6).
3889            if (329..=336).contains(&self.dot) {
3890                self.prefetch_shift_bg_regs();
3891            }
3892        }
3893        // Commit pending OAM corruption at the START of the pre-render line
3894        // (TriCNES handles this via the dots 1-64 eval path on the
3895        // pre-render line; the dot-0 hook covers the case where rendering
3896        // was re-enabled during VBlank and stays on into pre-render).
3897        // Per TriCNES `CorruptOAM`, the corruption applies on the first
3898        // rendered dot once rendering is (re-)enabled.
3899        if self.scanline == self.region.prerender_line()
3900            && self.dot == 0
3901            && rendering
3902            && self.oam_corruption_pending
3903        {
3904            self.process_oam_corruption();
3905        }
3906        // OAM corruption (TriCNES eval-pointer model): maintain the
3907        // `OAM2Address` analogue across the dots 1-64 clear window, capture
3908        // the corruption index at the disable edge, and commit on re-enable.
3909        // Driven every render-line dot independent of the rendering gate so
3910        // the disable edge is observed even though the sprite-eval FSM below
3911        // stops once rendering is gated off.
3912        if render_line {
3913            self.tick_oam_corruption(rendering);
3914        }
3915        self.prev_rendering_enabled = rendering;
3916        // v2.0 (ae30785): update the 1-dot-delayed copy AFTER this dot's gate
3917        // read above, so the next dot sees the delayed value.
3918        {
3919            #[cfg(feature = "phi2-write-sweep")]
3920            {
3921                self.render_gate_end_dot(render_gate_prev1);
3922                // Newest first: stage 0 is "one dot ago" on the next dot.
3923                self.sweep_mask_history.rotate_right(1);
3924                self.sweep_mask_history[0] = self.mask;
3925            }
3926            // Stage 2 takes stage 1 as it stood when this dot BEGAN, which is
3927            // exactly the `rendering_gate` local. Ordering is load-bearing in
3928            // the same way `render_gate_end_dot` is: reading the field back
3929            // here instead would assign stage 1 to stage 2 after stage 1 had
3930            // already been overwritten, freezing the pipeline -- the #506
3931            // defect in a second place.
3932            self.rendering_enabled_delayed2 = rendering_gate;
3933            self.rendering_enabled_delayed = rendering;
3934        }
3935
3936        // === Dot-1 / dot-0 events ===
3937        // VBL flag is set at scanline 241 dot 1 per nesdev wiki.  The
3938        // PPU's /NMI line is pulled low one PPU clock later (dot 2),
3939        // matching the behavior blargg's `ppu_vbl_nmi/05-nmi_timing` and
3940        // `08-nmi_off_timing` were calibrated to: a /NMI assertion-edge
3941        // sample ~6-7 PPU clocks after VBL set, given how our bus
3942        // interleaves CPU bus accesses before the 3-PPU-tick `on_cpu_cycle`
3943        // hook (vs. real hardware's mid-cycle phi1 access).
3944        #[cfg(not(feature = "phi2-write-sweep"))]
3945        let vbl_set_dot = 1u16;
3946        // v2.6.18 condition-2 diagnostic. NOT a proposed fix: nesdev states the
3947        // flag sets at scanline 241 dot 1, and this knob exists only to ask
3948        // whether the six NMI entries' failure at a later ACCESS dot is a pure
3949        // one-dot RELATIVE shift between the `$2002` read and VBL-set, or
3950        // something else. If shifting VBL by the same dot restores all six, the
3951        // disagreement is about CPU/PPU alignment rather than about the read.
3952        #[cfg(feature = "phi2-write-sweep")]
3953        let vbl_set_dot = u16::from(VBL_SET_DOT.load(core::sync::atomic::Ordering::Relaxed));
3954        if self.scanline == self.region.vblank_start_line()
3955            && self.dot == vbl_set_dot
3956            && !self.suppress_vbl_this_frame
3957        {
3958            self.status.insert(PpuStatus::VBLANK);
3959            // Inform the mapper that we have entered VBL — MMC5 uses this
3960            // to clear its in-frame flag.
3961            bus.notify_vblank();
3962            // R2 (mc-r1-substrate): /NMI is asserted on the SAME dot as
3963            // VBL-set when NMI_ENABLE is set (Mesen2 `NesPpu.cpp:1339-1343`).
3964            // On R1's on-time access the CPU read at dot 1 lands AFTER
3965            // VBL+NMI set, so the v1.x `dot==3` lag-comp band-aid (below) is
3966            // obsolete and disabled.
3967            if self.ctrl.contains(PpuCtrl::NMI_ENABLE) {
3968                self.nmi_line = true;
3969            }
3970        }
3971        // v1.x default path: /NMI raised at dot 3 to compensate for the
3972        // lockstep PPU running ~8 mc late. Disabled under the on-time R1
3973        // substrate (R2 raises /NMI at dot 1, above).
3974        if pre_render && self.dot == 1 {
3975            self.status.remove(
3976                PpuStatus::VBLANK
3977                    .union(PpuStatus::SPRITE_ZERO_HIT)
3978                    .union(PpuStatus::SPRITE_OVERFLOW),
3979            );
3980            self.nmi_line = false;
3981            self.suppress_vbl_this_frame = false;
3982        }
3983
3984        // Notify the mapper that a rendered scanline has started. We fire
3985        // this on dot 0 of every visible line and the pre-render line,
3986        // before any pattern/attribute fetches happen. MMC5 uses this to
3987        // tick its scanline IRQ counter (which conceptually fires at PPU
3988        // cycle ~4 of each rendered line — close enough for v0). Other
3989        // mappers default to no-op.
3990        if render_line && self.dot == 0 {
3991            bus.notify_scanline_start();
3992            // T-MMC3-BG-A12, rule 2: a visible line's dot 0 "appears to be
3993            // the same CHR address that is later used to fetch the low
3994            // background tile byte starting at dot 5" (`NESdev` PPU
3995            // rendering, "Cycle 0"), so with the background at `$1000` A12 is
3996            // already high here. Not on the pre-render line, which the page
3997            // does not extend the statement to and where giving it the rule
3998            // failed blargg `4-scanline_timing` at sub-test 8; and not on the
3999            // dot the odd-frame skip replaced (`dot0_replaced`), which is the
4000            // dummy nametable fetch's tick. The rule is what closes sub-test 12
4001            // (removing it returns both 4-scanline ROMs there); the second
4002            // exception is documented behaviour that no ROM in the corpus
4003            // reaches, pinned by the unit test
4004            // `scanline_0_dot_0_drives_bg_chr_only_when_the_skip_did_not_replace_it`.
4005            if self.mask.rendering_enabled()
4006                && !self.dot0_replaced
4007                && self.scanline != self.region.prerender_line()
4008            {
4009                let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4010                self.observe_a12_addr(bus, bg_table);
4011            }
4012            self.dot0_replaced = false;
4013        }
4014
4015        // === Background rendering pipeline (visible + pre-render lines) ===
4016        // v2.6.18 condition-4: when `SCROLL_GATE_LAG` is armed, the dot-256
4017        // vertical increment is evaluated HERE on its own `$2001` view.
4018        //
4019        // This is the consumer `Frozen OAM2 Increment` actually turns on. The
4020        // ROM enables rendering ON dot 256 and states outright that "the PPU's
4021        // vertical scroll is NOT incremented"; our enable lands a dot early, so
4022        // `inc_vert_v` fires and `v` ends at fine-Y 3 where the test needs 2 --
4023        // no opaque background pixel under the sprite, no sprite-zero hit, and
4024        // the entry fails for a reason that has nothing to do with OAM2.
4025        // `RENDER_GATE_LAG = 2` fixes it and breaks `Stale Sprite Shift Regs`,
4026        // because that depth moves every other consumer too.
4027        // v2.6.18: the dot-256 vertical increment reads rendering as of TWO
4028        // dots ago, not the shared one-dot gate -- see
4029        // `rendering_enabled_delayed2`. Hoisted out of the `render_line &&
4030        // rendering_gate` block so the shared gate cannot decide the question
4031        // before the deeper one is consulted; conjoining the two would put the
4032        // DISABLE edge back on the shallower gate, which is the edge
4033        // `Frozen OAM2 Increment` test 4 measures.
4034        if render_line
4035            && self.dot == 256
4036            && Self::scroll_gate_follows_render_gate()
4037            && rendering_gate2
4038        {
4039            self.inc_vert_v();
4040        }
4041        #[cfg(feature = "phi2-write-sweep")]
4042        if render_line && self.dot == 256 && !Self::scroll_gate_follows_render_gate() {
4043            let depth = SCROLL_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) as usize;
4044            let m = if depth == 0 {
4045                self.mask
4046            } else {
4047                self.sweep_mask_history[(depth - 1).min(self.sweep_mask_history.len() - 1)]
4048            };
4049            if m.rendering_enabled() {
4050                self.inc_vert_v();
4051            }
4052        }
4053        // v2.6.18: when `SPRITE_REARM_LAG` is armed the dot-339 re-arm is
4054        // evaluated HERE, outside the shared `rendering_gate` block, so it can
4055        // see a rendering value the shared gate does not. Hoisted rather than
4056        // gated in place because the enclosing block would otherwise decide the
4057        // question before the knob is consulted.
4058        #[cfg(feature = "phi2-write-sweep")]
4059        if render_line && self.dot == 339 && !Self::sprite_rearm_follows_render_gate() {
4060            let depth = SPRITE_REARM_LAG.load(core::sync::atomic::Ordering::Relaxed) as usize;
4061            let m = if depth == 0 {
4062                self.mask
4063            } else {
4064                self.sweep_mask_history[(depth - 1).min(self.sweep_mask_history.len() - 1)]
4065            };
4066            if m.rendering_enabled() {
4067                for i in 0..self.spr_count as usize {
4068                    self.spr_halted[i] = false;
4069                }
4070            }
4071        }
4072        if render_line && rendering_gate {
4073            // Sprite evaluation: per-PPU-dot FSM matching real-hardware
4074            // behavior (cycles 1-64 secondary-OAM clear, 65-256 alternating
4075            // odd/even read/write with the documented buggy `n+m`
4076            // overflow-detection increment).  Visible scanlines evaluate
4077            // for the next visible scanline; the pre-render line evaluates
4078            // for scanline 0.  Without the pre-render eval, secondary OAM
4079            // from the last visible scanline would leak into pre-render's
4080            // dummy sprite tile fetches, causing wrong A12 emissions and
4081            // incorrect sprite-zero state.
4082            if visible || pre_render {
4083                self.tick_sprite_eval_per_dot();
4084            }
4085            // v2.0 Tier 1.2: drive the isolated OAM-data-bus model on visible
4086            // scanlines when rendering, so a CPU $2004 read mid-frame observes
4087            // the sprite-eval / load data bus (AccuracyCoin `$2004 Stress`).
4088            // Side-effect-free w.r.t. the rendering FSM above.
4089            // v2.6.18: the OAM2 machinery's effective gate is a CONJUNCTION --
4090            // this test AND the enclosing `render_line && rendering_gate`. That
4091            // matters and is easy to state wrongly: for an AND of two delayed
4092            // views of one signal, the DISABLE edge fires at the shallower depth
4093            // and the RE-ENABLE edge at the deeper one. So today
4094            // `OAM2_GATE_LAG = 0` owns the disable edge and the 1-dot
4095            // `rendering_enabled_delayed` owns the re-enable edge -- the two
4096            // knobs each own ONE edge, which is why neither alone can place the
4097            // window and why `Frozen OAM2 Increment` moves only when
4098            // `RENDER_GATE_LAG` does.
4099            //
4100            // TriCNES has no such nesting: `PPU_Render_SpriteEvaluation()` is
4101            // called unconditionally and each block tests exactly one mask view
4102            // (`Emulator.cs:2073`, `:2076-2082`).
4103            //
4104            // `OAM2_GATE_LAG` makes this test's depth swept rather than assumed;
4105            // 0 is the shipped live-mask read of THIS term, not of the gate.
4106            // v2.6.18: this used to conjoin the LIVE mask, which put the OAM2
4107            // counter's DISABLE edge a dot ahead of every other consumer --
4108            // `min(r, d)` for a conjunction of two delayed views. The dot-339
4109            // reset then got skipped by a write whose effect lands DURING 339,
4110            // leaving the "OAM2 Overflowed" flag raised and producing the
4111            // sprite-zero hit `Frozen OAM2 Increment` test 4 exists to forbid.
4112            // The counter rides the shared one-dot gate like everything else,
4113            // which the enclosing block has already applied.
4114            if visible && self.oam2_gate_rendering() {
4115                self.tick_oam_bus();
4116            }
4117            // Sprite tile fetch + A12 emission.  Real hardware spreads the
4118            // 8 sprite slots' pattern fetches across cycles 257..=320 — for
4119            // each slot, garbage NT bytes at +1/+3, sprite pattern lo at
4120            // +5/+6, sprite pattern hi at +7/+8.  We collapse that to per-
4121            // slot emission at dot 260 for slot 0, 268 for slot 1, …, 316
4122            // for slot 7.  This is the canonical "MMC3 IRQ at PPU dot 260"
4123            // timing — the first A12 rise to the sprite pattern table
4124            // happens here for standard pattern-table layout (BG=$0000,
4125            // sprites=$1000), per `docs/mappers.md` §MMC3 → IRQ counter
4126            // mechanism.
4127            //
4128            // CRITICAL for MMC3: even unused sprite slots ALWAYS perform
4129            // the dummy sprite-pattern fetch on real hardware (using the
4130            // cleared secondary-OAM tile $FF), so A12 toggles into the
4131            // sprite pattern table once per scanline regardless of how
4132            // many real sprites are visible.  This must run on both
4133            // visible scanlines and the pre-render line — pre-render
4134            // sprite fetches are for scanline 0's sprites and contribute
4135            // the 241st A12 rising edge per frame (240 visible + 1
4136            // pre-render) that MMC3's IRQ counter expects.
4137            if (260..=316).contains(&self.dot) {
4138                let phase = self.dot.wrapping_sub(260);
4139                if phase.trailing_zeros() >= 3 {
4140                    let slot = (phase >> 3) as usize;
4141                    self.fetch_sprite_tile(bus, slot);
4142                }
4143            }
4144
4145            // OAMADDR reset: per nesdev wiki "PPU registers" §OAMADDR,
4146            // "OAMADDR is set to 0 during each of ticks 257-320 (the
4147            // sprite tile loading interval) of the pre-render and visible
4148            // scanlines." This is the hardware behaviour that lets games
4149            // STX $4014 their OAM-staging page after rendering without
4150            // having to remember to STA $2003 #0 first. Required for
4151            // AccuracyCoin TEST_Sprite0Hit_Behavior subtest 1 (which
4152            // relies on the prior test-runner's OAMADDR perturbation
4153            // being washed away by the previous frame's rendering).
4154            if (257..=320).contains(&self.dot) {
4155                self.oam_addr = 0;
4156            }
4157
4158            // v2.0 (ppu-sprite-shifter-counter): at dot 339 of a render line,
4159            // re-arm the LOADED sprite counters (the `spr_count` slots fetched
4160            // for the next scanline) to "counting". This whole block is gated on
4161            // `rendering_gate`, so a render-disable across dot 339 leaves the
4162            // loaded slots halted — a reloaded-but-halted counter draws
4163            // immediately on re-enable (Stale Sprite Shift Regs t5/6). Slots
4164            // beyond `spr_count` retain their halted latch.
4165            if self.dot == 339 && Self::sprite_rearm_follows_render_gate() {
4166                for i in 0..self.spr_count as usize {
4167                    self.spr_halted[i] = false;
4168                }
4169            }
4170
4171            // BG fetches happen at dots 1..=256 and 321..=336.
4172            //
4173            // CYCLE-PRECISE BG PIPELINE (Mesen2-faithful, fixes Cascade A
4174            // VerifySpriteZeroHits step-2 off-by-one):
4175            //
4176            // Per nesdev wiki "PPU rendering": "The shifters are reloaded
4177            // during ticks 9, 17, 25, ..., 257." Per Mesen2
4178            // `Core/NES/NesPpu.cpp::LoadTileInfo()` (line 667), the reload
4179            // is `case 1` of `(_cycle & 0x07)` — i.e., phase 0 of each
4180            // 8-cycle group, OR'ing the latched LowByte/HighByte into the
4181            // shifter's low 8 bits. The PRIOR group's 8 shifts (one per
4182            // cycle of dots 1..=256 of a visible scanline) leave bits 0-7
4183            // zeroed, so the OR is effectively an overwrite. The pre-fetch
4184            // groups at dots 321..=336 do NOT shift per-cycle; instead
4185            // Mesen2 substitutes a `<<= 8` at phase 7 (dots 328 and 336)
4186            // to clear bits 0-7 for the next reload.
4187            //
4188            // The matching pixel-emit + shift ordering is: emit_pixel reads
4189            // bit (15 - fine_x) FIRST, then shift_bg runs LAST. This is the
4190            // critical change from the prior (off-by-one) implementation
4191            // that shifted BEFORE emit and reloaded at phase 7 (cycle 8).
4192            // See `docs/audit/cascade-a-investigation-2026-05-19.md` for
4193            // the empirical analysis and the per-cycle trace of
4194            // VerifySpriteZeroHits step 2 demonstrating why this is the
4195            // load-bearing change.
4196            let in_bg_fetch = (1..=256).contains(&self.dot) || (321..=336).contains(&self.dot);
4197            if in_bg_fetch {
4198                let phase = (self.dot.wrapping_sub(1)) & 7;
4199                // Phase 0 (cycles 1, 9, 17, ..., 249, 321, 329): reload the
4200                // shifter's low 8 bits from the latches written by the
4201                // PRIOR fetch group. Implementation note: `reload_bg_shift_regs`
4202                // overwrites bits 0-7 via `(shift & 0xFF00) | latch`; this
4203                // matches Mesen2's `|=` because the 8 shifts since the prior
4204                // reload guarantee bits 0-7 are zero before the OR.
4205                if phase == 0 {
4206                    // v2.0 Phase 6 (mc-ppu-subpos): the reload is additionally
4207                    // gated on the analog-delayed `$2001` value, so a render
4208                    // re-enable lands the reload `MASK_WRITE_DELAY` dots after
4209                    // the shifter has already resumed advancing -> one reload is
4210                    // SKIPPED and the serial-in '1's survive (BG Serial In).
4211                    // When stable, `bg_reload_render` == the live rendering gate,
4212                    // so this is byte-identical.
4213                    let reload_gate = self.bg_reload_render;
4214                    if reload_gate {
4215                        self.reload_bg_shift_regs();
4216                    }
4217                }
4218
4219                // v2.0.3 (ADR 0030, Option 1) — the ALE (address-latch-enable)
4220                // half of each 2-cycle VRAM access. On the EVEN dot of each pair
4221                // (phases 0/2/4/6, i.e. one dot BEFORE the corresponding read at
4222                // phases 1/3/5/7) the PPU drives the full 14-bit fetch address onto
4223                // `address_bus` and captures its low byte into `octal_latch`; the
4224                // read half (the existing `fetch_*` below) reads via the splice
4225                // `(address_bus & 0x3F00) | octal_latch`. For a coherent fetch the
4226                // splice returns the intended address, so this is behavior-neutral;
4227                // the split exists so a later phase's `$2006`/`$2007` corruption can
4228                // desync the two halves naturally.
4229                match phase {
4230                    // Phase 0 (NT ALE): drive the PLAIN nametable address
4231                    // (`0x2000 | (v & 0x0FFF)`) and load `octal_latch` with its low
4232                    // byte. This is a TRUE two-dot ALE for the common (non-MMC5-
4233                    // split) case, so the latch naturally carries the NT-low the
4234                    // "Hybrid Addresses" corruption needs. The MMC5 vertical-split
4235                    // query stays at the read dot (phase 1, `fetch_nt`) because its
4236                    // `split_chr_bank_latch` side effect is mapper-observable; when
4237                    // that query turns out to be split-active, `fetch_nt` disarms
4238                    // this ALE and reads `split.nt_addr` co-located instead, so
4239                    // split rendering is byte-identical (no phase-0 mapper query).
4240                    0 => self.ale_drive_nt(),
4241                    2 => self.ale_drive_at(),
4242                    4 => self.ale_drive_bg_lo(),
4243                    6 => self.ale_drive_bg_hi(),
4244                    _ => {}
4245                }
4246
4247                // 8-cycle fetch group: dot phase = (dot - 1) & 7
4248                //   1 -> NT byte fetch     (cycle 2 of group)
4249                //   3 -> AT byte fetch     (cycle 4 of group)
4250                //   5 -> BG-low fetch      (cycle 6 of group)
4251                //   7 -> BG-high fetch +
4252                //        coarse-X increment (cycle 8 of group)
4253                match phase {
4254                    1 => self.fetch_nt(bus),
4255                    3 => self.fetch_at(bus),
4256                    5 => self.fetch_bg_lo(bus),
4257                    7 => self.fetch_bg_hi(bus),
4258                    _ => {}
4259                }
4260                self.observe_bg_a12_lead(bus, phase);
4261                if phase == 7 {
4262                    self.inc_hori_v();
4263                    // Pre-fetch region only (dots 328 and 336): explicit
4264                    // `<<= 8` to substitute for the missing per-cycle
4265                    // shifts during pre-fetch. Per Mesen2
4266                    // `ProcessScanlineImpl()` lines 941-944.
4267                    if (321..=336).contains(&self.dot) {
4268                        self.prefetch_shift_bg_regs();
4269                    }
4270                }
4271            }
4272            // Dot 257: the LAST shift-register reload of the visible region
4273            // consumes the latches from the dots-249..=256 fetch group. Dot
4274            // 257 is outside the dots 1..=256 `in_bg_fetch` range above (it
4275            // belongs to the sprite-tile-fetch window 257..=320), but per
4276            // Mesen2's `_cycle <= 256` LoadTileInfo cycle range, this reload
4277            // actually never fires in Mesen2 either — the dot-256 fetch's
4278            // bg_lo/bg_hi latches are consumed by the dot-321 reload (which
4279            // OR's them in, then the dot-328 `<<= 8` shifts them up to
4280            // bits 8-15). So for our model, the dot-256 fetch's latches
4281            // similarly persist past dot 256 into dot 321's reload.
4282            // (Intentionally no dot-257 reload here.)
4283
4284            // Cycle 256: vertical-V increment -- HOISTED to the top of the
4285            // dot-event section (search `rendering_enabled_delayed2`), because
4286            // it reads a rendering gate one dot deeper than the shared one and
4287            // must therefore not sit inside the shared gate's block.
4288            // Cycle 257: copy hori(t) -> hori(v).
4289            if self.dot == 257 {
4290                // W2 ($2007 Stress): the FIRST garbage NT read of sprite slot
4291                // 0 has its ALE at dot 257, BEFORE hori(v) is reset — so its
4292                // ADDRESS uses the OLD `v` (coarse-x wrapped past the row's
4293                // last column) even though the data lands at dot 258. Latch
4294                // the address here; `tick_sprite_fetch_read` uses it for the
4295                // slot-0 first read (key idx 128 = `02`). The later garbage
4296                // NT reads (ALE dots 259+) all use the reset `v`.
4297                {
4298                    self.ppudata_spr0_nt_addr = 0x2000 | (self.v & 0x0FFF);
4299                }
4300                self.copy_hori_t_to_v();
4301            }
4302            // Pre-render cycles 280..=304: copy vert(t) -> vert(v).
4303            if pre_render && (280..=304).contains(&self.dot) {
4304                self.copy_vert_t_to_v();
4305            }
4306            // Sprite tile fetch happens in fetch_sprite_tile (dots 260, 268, ..., 316).
4307            // Cycles 337..=340: 2 garbage NT fetches (no-op except A12).
4308            if (337..=340).contains(&self.dot) && (self.dot & 1) == 1 {
4309                self.fetch_nt(bus);
4310            }
4311
4312            // W2 ($2007 Stress): the per-dot sprite-fetch read cadence (dots
4313            // 257-320) feeds `render_data_bus` (NT,NT,PT-lo,PT-hi per 8-dot
4314            // slot) so a deferred `$2007` buffer reload landing in HBlank
4315            // captures the byte the sprite fetch drove on the VRAM bus. The
4316            // real fetch is the collapsed `fetch_sprite_tile` above; this is
4317            // side-effect-free w.r.t. rendering and A12 (the garbage NT reads
4318            // are CIRAM/nametable reads; the PT bytes come from the stash).
4319            if (257..=320).contains(&self.dot) {
4320                self.tick_sprite_fetch_read(bus);
4321            }
4322        }
4323
4324        // W2 ($2007 Stress): the PPUDATA state machine's read step (PD_RB) +
4325        // TStep. A `$2007` read during rendering armed this countdown; one
4326        // tick per PPU dot (unconditionally, so a mid-flight rendering
4327        // disable cannot wedge it), and at 0:
4328        //   1. `data_buffer` latches the value the FETCH cadence drove on the
4329        //      VRAM bus at THIS dot (`render_data_bus`, freshly set by the
4330        //      fetch dispatch above). Latched bus value only — never a fresh
4331        //      VRAM read (zero new A12/mapper events).
4332        //   2. The deferred v-glitch increment (the TStep) fires AFTER the
4333        //      reload, per TriCNES `PPU_DATA_StateMachine_Half` — so every
4334        //      fetch in the read-to-reload window used the OLD `v`. The
4335        //      rendering-vs-blanking choice uses the state AT the TStep dot
4336        //      (mirrors TriCNES's `PPU_2007_BLNK_Latch`).
4337        if self.ppudata_sm_countdown > 0 {
4338            self.ppudata_sm_countdown -= 1;
4339            if self.ppudata_sm_countdown == 0 {
4340                self.data_buffer = self.render_data_bus;
4341                // v2.0.2 (ADR 0030) — "ALE + Read": the $2007 read's PPUDATA
4342                // state machine takes 3 PPU cycles to the background cadence's 2,
4343                // so on the reload dot its ALE overlaps a fetch's read. Both ALE
4344                // and READ are asserted, so the octal latch is frozen on the
4345                // read's DATA byte (`render_data_bus`) instead of the next
4346                // Pattern-Address-Register low byte. The next pattern fetch then
4347                // reads `{PAR high 6}:{stale data low}` (`$0F03` -> `$0FFF`).
4348                if self.mask.rendering_enabled() && self.is_render_scanline() {
4349                    // Freeze the latch on the read's DATA byte here. The frozen byte
4350                    // is then carried by `drive_bus` (which suppresses the next
4351                    // pattern ALE's latch reload) and consumed by the pattern read's
4352                    // natural `ale_splice`, so the next pattern fetch reads
4353                    // `(PAR high 6):(stale $FF) = $0FFF` with no explicit splice.
4354                    self.octal_latch = self.render_data_bus;
4355                    self.pattern_latch_stale = true;
4356                    octal_trace::push(
4357                        octal_trace::K_SMLAND,
4358                        self.frame,
4359                        self.scanline,
4360                        self.dot,
4361                        u32::from(self.render_data_bus),
4362                    );
4363                }
4364                if self.ppudata_v_inc_pending {
4365                    self.ppudata_v_inc_pending = false;
4366                    if self.mask.rendering_enabled() && self.is_render_scanline() {
4367                        self.inc_hori_v();
4368                        self.inc_vert_v();
4369                    } else {
4370                        let inc = if self.ctrl.contains(PpuCtrl::VRAM_INCREMENT_32) {
4371                            32
4372                        } else {
4373                            1
4374                        };
4375                        self.v = self.v.wrapping_add(inc) & 0x7FFF;
4376                    }
4377                    // The v change can move A12 (mirrors the read-time path).
4378                    self.observe_a12(bus);
4379                }
4380            }
4381        }
4382
4383        // === Pixel emission (visible scanlines, dots 1..=256) ===
4384        // Per Mesen2 `ProcessScanlineImpl()` (lines 881-884), the
4385        // canonical order is: LoadTileInfo (reload at phase 0, fetches at
4386        // phases 1/3/5/7) THEN DrawPixel THEN ShiftTileRegisters. The
4387        // shift-AFTER-emit ordering is the load-bearing other half of the
4388        // Cascade A BG-pipeline fix: emit reads bit (15 - fine_x) of the
4389        // shifter at its CURRENT state (post-reload, pre-shift), then the
4390        // shift advances the register for the next emit. Combined with
4391        // the phase-0 reload above, this places the newly-fetched tile's
4392        // MSB at shift-register bit 15 (the emit read point) at exactly
4393        // PPU dot 9 of each 8-cycle group = pixel column 8.
4394        if visible && (1..=256).contains(&self.dot) {
4395            self.emit_pixel();
4396            // v2.0 (ae30785): the BG shifter advances on the 1-dot-delayed
4397            // rendering gate (under the feature), so a precisely-timed `$2001`
4398            // toggle that skips the reload while rendering stays enabled
4399            // surfaces the serial-in (BG Serial In / Stale BG Shift).
4400            //
4401            // v2.0 Phase 6 (mc-ppu-subpos): TriCNES drives the SHIFT off the
4402            // IMMEDIATE PPUMASK (`_EmulateHalfPPU` -> `PPU_UpdateBackground
4403            // ShiftRegisters`, gated on `PPU_Mask_Show*` not `*_Delayed`) while
4404            // the fetch+RELOAD ride the 1-dot-DELAYED mask
4405            // (`PPU_Render_ShiftRegistersAndBitPlanes`, gated on `*_Delayed`).
4406            // So on a render re-enable edge there is a 1-dot window where the
4407            // shifter advances (injecting the serial-in '1') but the reload is
4408            // still gated off -> a single reload is SKIPPED while shifting
4409            // continues, surfacing the accumulated serial-in '1's at the output
4410            // (AccuracyCoin "BG Serial In"). When no mid-scanline toggle is in
4411            // flight immediate == delayed, so normal rendering is byte-identical.
4412            let shift_gate = render_line && rendering;
4413            if shift_gate {
4414                self.shift_bg();
4415            }
4416        }
4417
4418        // === Per-PPU-dot state-trace recording (Session-10) ===
4419        //
4420        // Gated on the `ppu-state-trace` cargo feature so the
4421        // default build's hot tick path is byte-identical to
4422        // pre-Session-10. The hook reads `self`'s state AFTER
4423        // all this dot's effects have applied, so the captured
4424        // record reflects "PPU state at the end of dot
4425        // (scanline, dot)". It NEVER writes to PPU state — the
4426        // determinism contract is preserved.
4427        //
4428        // See `docs/adr/0005-ppu-state-trace.md`.
4429        #[cfg(feature = "ppu-state-trace")]
4430        if self.state_trace.is_some() {
4431            let rec = self.build_state_record();
4432            if let Some(t) = self.state_trace.as_mut() {
4433                t.maybe_push(rec);
4434            }
4435        }
4436    }
4437
4438    /// v2.1.8 A1 — the specialized straight-line body for a "clean" visible
4439    /// BG-render dot: a visible scanline, `dot` in `1..=256`, rendering stably
4440    /// enabled, and no sub-dot disturbance in flight. Dispatched from
4441    /// [`Self::tick`] behind the [`Self::fast_dotloop`] guard (default ON
4442    /// since v2.2.3).
4443    ///
4444    /// This executes the *exact same* helper sequence the general per-dot path
4445    /// runs for such a dot — in the same order — with every event and
4446    /// bookkeeping branch the guard proves un-taken (VBL/NMI set/clear, the
4447    /// pre-render vertical reload, sprite-tile fetch dots 260..=316, the
4448    /// OAMADDR-reset window, the dot-257 hori-copy, the PPUDATA state machine,
4449    /// the OAM-corruption commit, the odd-frame skip) elided. It is therefore
4450    /// byte-identical to the general path by construction, and is additionally
4451    /// pinned bit-for-bit by the differential test + the full oracle. See the
4452    /// extensive rationale at the dispatch site in [`Self::tick`].
4453    /// v2.2.3 P2 — the specialized straight-line body for an **idle-line** dot:
4454    /// a post-render or vblank line other than the VBL-set line, with no
4455    /// sub-dot disturbance in flight. Dispatched from [`Self::tick`] behind the
4456    /// [`Self::fast_dotloop`] guard.
4457    ///
4458    /// An idle line issues no VRAM fetch, emits no pixel, runs no sprite
4459    /// evaluation, clocks no shifter, and raises no VBL / NMI / A12 /
4460    /// scanline-start event. Walking the general per-dot body for it therefore
4461    /// evaluates ~30 predicates to perform three assignments. This is those
4462    /// three assignments, in the general path's order, derived from one
4463    /// `mask.rendering_enabled()` read exactly as it does:
4464    ///
4465    /// 1. `bg_reload_render` — the general path's `mask_write_delay` `else`
4466    ///    arm. The guard proves the countdown is zero, so that arm is the one
4467    ///    taken.
4468    /// 2. `prev_rendering_enabled` — assigned unconditionally there.
4469    /// 3. `rendering_enabled_delayed` — likewise, and deliberately AFTER (2):
4470    ///    the general path updates the 1-dot-delayed copy last so the *next*
4471    ///    dot observes it, and reordering here would shift a mid-vblank
4472    ///    `$2001` toggle by one dot.
4473    ///
4474    /// Byte-identical by construction, and pinned bit-for-bit by
4475    /// `fast_dotloop_diff` (which compares whole frames — every idle dot
4476    /// included — and by `idle_line_fast_path_matches_exact_under_vblank_io`,
4477    /// which drives `$2000`/`$2001`/`$2006`/`$2007` during vblank so the
4478    /// guard's fall-through arms are exercised rather than assumed).
4479    #[cfg(feature = "ppu-idle-line-fast")]
4480    #[inline]
4481    const fn tick_idle_line_fast(&mut self) {
4482        let rendering = self.mask.rendering_enabled();
4483        self.bg_reload_render = rendering;
4484        self.prev_rendering_enabled = rendering;
4485        // Stage 2 before stage 1, for the same ordering reason as the general
4486        // path: stage 1's PREVIOUS value is what stage 2 takes.
4487        self.rendering_enabled_delayed2 = self.rendering_enabled_delayed;
4488        self.rendering_enabled_delayed = rendering;
4489    }
4490
4491    // Dead under `ppu-state-trace`, and legitimately so: the dispatch above is
4492    // `#[cfg(not(feature = "ppu-state-trace"))]`, because the trace hook must
4493    // observe EVERY dot and the fast path exists precisely to skip per-dot work.
4494    // Live by default, dead under one feature -- the one shape that earns an
4495    // `allow` rather than a deletion. Scoped to that feature so the attribute
4496    // cannot silently start suppressing a real finding in the default build.
4497    #[cfg_attr(feature = "ppu-state-trace", allow(dead_code))]
4498    #[inline]
4499    fn tick_visible_render_fast<B: PpuBus>(&mut self, bus: &mut B) {
4500        let dot = self.dot;
4501
4502        // General-path top: with `mask_write_delay == 0` (guard) the BG-reload
4503        // gate follows the stably-enabled rendering bit.
4504        self.bg_reload_render = true;
4505
4506        // OAM-corruption pointer bookkeeping. The guard proved nothing is
4507        // armed/pending/disabled, so this only maintains `oam2_addr` across the
4508        // dots 1..=64 secondary-OAM clear window (and is a two-compare no-op for
4509        // dots 65..=256) — exactly what the general path's
4510        // `if render_line { tick_oam_corruption(rendering) }` does here.
4511        self.tick_oam_corruption(true);
4512
4513        // Rendering-edge bookkeeping the NEXT dot's gate consumes. The general
4514        // path assigns all three every dot (`delayed2 <- delayed <- rendering`,
4515        // `prev <- rendering`); here the guard in `tick` has already required
4516        // `prev_rendering_enabled`, `rendering_enabled_delayed`,
4517        // `rendering_enabled_delayed2` and `mask.rendering_enabled()`, so each
4518        // assignment would write the value the field already holds, and the
4519        // state a fast→general dot boundary hands on is the same either way.
4520        //
4521        // Until v2.7.6 this block re-wrote them to `true`. Stated as the
4522        // invariant instead, so a future change to the guard that stops
4523        // guaranteeing it fails here in every debug and test build rather than
4524        // having the stores silently paper over it. Measured on its own in
4525        // v2.7.6 (core audit IMP-06, `docs/performance.md`): deleting the
4526        // stores is byte-identical and moves nothing, so this is a statement of
4527        // the invariant, not an optimisation.
4528        debug_assert!(
4529            self.prev_rendering_enabled
4530                && self.rendering_enabled_delayed
4531                && self.rendering_enabled_delayed2,
4532            "fast render path entered without a stably-enabled rendering history"
4533        );
4534
4535        // Sprite-evaluation FSM (visible scanline) + isolated OAM data-bus model.
4536        self.tick_sprite_eval_per_dot();
4537        self.tick_oam_bus();
4538
4539        // Background fetch pipeline: dots 1..=256 are all in the fetch window,
4540        // `phase = (dot - 1) & 7`.
4541        let phase = dot.wrapping_sub(1) & 7;
4542        // Phase 0: shift-register reload (reload gate == `bg_reload_render`).
4543        if phase == 0 {
4544            self.reload_bg_shift_regs();
4545        }
4546        // 2-cycle-ALE address-latch half (even phases).
4547        match phase {
4548            0 => self.ale_drive_nt(),
4549            2 => self.ale_drive_at(),
4550            4 => self.ale_drive_bg_lo(),
4551            6 => self.ale_drive_bg_hi(),
4552            _ => {}
4553        }
4554        // Read half (odd phases).
4555        match phase {
4556            1 => self.fetch_nt(bus),
4557            3 => self.fetch_at(bus),
4558            5 => self.fetch_bg_lo(bus),
4559            7 => self.fetch_bg_hi(bus),
4560            _ => {}
4561        }
4562        self.observe_bg_a12_lead(bus, phase);
4563        // Phase 7 (cycle 8 of the group): coarse-X increment. The dots
4564        // 321..=336 prefetch `<<= 8` is out of the 1..=256 range, so it never
4565        // applies here.
4566        if phase == 7 {
4567            self.inc_hori_v();
4568        }
4569        // Dot 256: vertical-V increment (with the 29→0 wrap-and-flip quirk).
4570        if dot == 256 {
4571            self.inc_vert_v();
4572        }
4573
4574        // Pixel emission + BG shift. The shift gate `render_line && rendering`
4575        // is `true` throughout the covered window.
4576        self.emit_pixel();
4577        self.shift_bg();
4578    }
4579
4580    // ------------------------------------------------------------------
4581    // Background fetch + shift + increment helpers.
4582    // ------------------------------------------------------------------
4583
4584    /// Fetch the nametable byte for the current `v`. Address: `$2000 |
4585    /// (v & 0x0FFF)`.
4586    ///
4587    /// MMC5 vertical split-screen: at the boundary of each 8-dot BG fetch
4588    /// group, the mapper is consulted via `bus.bg_split_state(...)`. If
4589    /// the current tile column falls within the alt region, the returned
4590    /// state supplies the synthesized NT / AT addresses, the alt fine-Y,
4591    /// and the 4 KiB CHR bank index. We latch it onto `bg_split_latch` for
4592    /// consumption by AT / BG-lo / BG-hi within the same fetch group.
4593    #[allow(clippy::cast_sign_loss)]
4594    #[inline]
4595    fn fetch_nt<B: PpuBus>(&mut self, bus: &mut B) {
4596        // Compute the (scanline_y, coarse_x) the alt region would be sampled
4597        // at. The pre-render line passes 0 (the alt region only renders on
4598        // visible lines, but the query is benign for pre-render).
4599        let scanline_y = if self.scanline < 0 {
4600            0
4601        } else {
4602            self.scanline as u16
4603        };
4604        let coarse_x = self.v & 0x001F;
4605        // NOTE (v2.0.3 / ADR 0030, Option 1): the MMC5 vertical-split query stays
4606        // HERE at the read dot — its `split_chr_bank_latch` side effect (which
4607        // `nametable_fetch`/`chr_offset` read) is mapper-observable, and moving it
4608        // one dot earlier to a phase-0 ALE shifts the Uchuu Keibitai SDF split
4609        // rendering. Consequently the NT fetch's octal-latch load co-locates with
4610        // its read (via `ale_splice`'s not-armed path below) rather than a phase-0
4611        // ALE; the AT / pattern fetches ARE true two-dot ALEs. See the plan.
4612        self.bg_split_latch = bus.bg_split_state(scanline_y, coarse_x);
4613
4614        let nt_addr = if let Some(split) = self.bg_split_latch {
4615            split.nt_addr
4616        } else {
4617            0x2000 | (self.v & 0x0FFF)
4618        };
4619        // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half. For the common
4620        // (non-split) case the phase-0 NT ALE already drove the plain address and
4621        // loaded `octal_latch`, so `ale_splice` takes its armed path and the read
4622        // address is the true multiplexed splice `(address_bus & 0x3F00) |
4623        // octal_latch` — transparent for a coherent fetch, and the divergence
4624        // point for the delayed-`CopyV` "Hybrid Addresses" corruption. For an
4625        // MMC5-split fetch the phase-0 ALE drove the PLAIN address (the split
4626        // query lives HERE for its mapper-observable side effect), so disarm and
4627        // read `split.nt_addr` co-located instead — byte-identical to a coherent
4628        // fetch.
4629        let nt_addr = {
4630            if self.bg_split_latch.is_some() {
4631                self.ale_armed = false;
4632            }
4633            self.ale_splice(nt_addr)
4634        };
4635        self.nt_latch = self.read_vram(bus, nt_addr);
4636        // v2.3.2 "Lucid" — capture the address this tile's number came from. The
4637        // SPLICED address, i.e. the one actually driven, so a hybrid-address
4638        // corruption shows the address the hardware really read rather than the
4639        // one it meant to. Telemetry only.
4640        #[cfg(feature = "debug-hooks")]
4641        {
4642            self.prov_nt_pending = nt_addr;
4643        }
4644        // Data phase: drive the byte just read back onto the multiplexed bus's low
4645        // 8 bits (AD7-0). Behavior-neutral (the next fetch's ALE overwrites it).
4646        self.ale_drive_data(self.nt_latch);
4647        // Latch any per-tile extended-attribute info (MMC5 ExGrafix). Skip
4648        // when split is active: the alt region uses standard 4-bit AT
4649        // semantics, not ExGrafix.
4650        self.ex_attr_latch = if self.bg_split_latch.is_some() {
4651            None
4652        } else {
4653            bus.peek_ex_attribute(self.v)
4654        };
4655    }
4656
4657    /// Fetch the attribute byte for the current `v`. Address:
4658    /// `$23C0 | (v & 0x0C00) | ((v >> 4) & 0x38) | ((v >> 2) & 0x07)`.
4659    #[inline]
4660    fn fetch_at<B: PpuBus>(&mut self, bus: &mut B) {
4661        // Split active: use the alt AT address and recover coarse-X / coarse-Y
4662        // from the latched split state's NT address (where coarse-X = bits
4663        // 0..=4, coarse-Y = bits 5..=9).
4664        if let Some(split) = self.bg_split_latch {
4665            let at_addr = split.at_addr;
4666            // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half (split AT path):
4667            // splice / consume the ALE arm so it can't leak to the next fetch.
4668            let at_addr = self.ale_splice(at_addr);
4669            let byte = self.read_vram(bus, at_addr);
4670            // v2.3.2 "Lucid" — the split's own attribute address, which the
4671            // standard `$23C0 | ...` arithmetic cannot reproduce. This branch is
4672            // exactly why the record carries `at` instead of deriving it.
4673            #[cfg(feature = "debug-hooks")]
4674            {
4675                self.prov_at_pending = at_addr;
4676            }
4677            self.ale_drive_data(byte);
4678            let coarse_x = (split.nt_addr & 0x001F) as u8;
4679            let coarse_y = ((split.nt_addr >> 5) & 0x001F) as u8;
4680            let shift = ((coarse_y & 0x02) << 1) | (coarse_x & 0x02);
4681            self.at_latch = (byte >> shift) & 0x03;
4682            return;
4683        }
4684        let v = self.v;
4685        let at_addr = 0x23C0 | (v & 0x0C00) | ((v >> 4) & 0x38) | ((v >> 2) & 0x07);
4686        // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half (normal AT path).
4687        let at_addr = self.ale_splice(at_addr);
4688        let byte = self.read_vram(bus, at_addr);
4689        // v2.3.2 "Lucid" — see the split branch above.
4690        #[cfg(feature = "debug-hooks")]
4691        {
4692            self.prov_at_pending = at_addr;
4693        }
4694        self.ale_drive_data(byte);
4695        // Pick the 2-bit attribute based on coarse-X[1] and coarse-Y[1].
4696        let coarse_x = (v & 0x1F) as u8;
4697        let coarse_y = ((v >> 5) & 0x1F) as u8;
4698        let shift = ((coarse_y & 0x02) << 1) | (coarse_x & 0x02);
4699        let standard_palette = (byte >> shift) & 0x03;
4700        // ExGrafix override: replace the 2-bit palette with the per-tile
4701        // value latched at NT-fetch time.
4702        self.at_latch = self
4703            .ex_attr_latch
4704            .map_or(standard_palette, |ex| ex.palette & 0x03);
4705    }
4706
4707    /// Fetch BG pattern low byte for the current `nt_latch` + fine-Y of `v`.
4708    ///
4709    /// In MMC5 `ExGrafix` mode the mapper has internally latched a per-tile
4710    /// 4 KiB CHR bank from the most recent `peek_ex_attribute` call; it
4711    /// will resolve this `addr` against that bank rather than the standard
4712    /// BG bank registers. No address-bus rerouting required.
4713    ///
4714    /// In MMC5 vertical split-screen mode the mapper has likewise latched
4715    /// the `$5202` 4 KiB CHR bank from the most recent `bg_split_state`
4716    /// call, and the alt fine-Y replaces `v`'s fine-Y.
4717    #[inline]
4718    fn fetch_bg_lo<B: PpuBus>(&mut self, bus: &mut B) {
4719        let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4720        let fine_y = self
4721            .bg_split_latch
4722            .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4723        let addr = bg_table | (u16::from(self.nt_latch) << 4) | fine_y;
4724        self.observe_a12_addr(bus, addr);
4725        // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half: A12 above stays on the
4726        // INTENDED `addr`; only the DATA read address goes through the ALE splice
4727        // (stale-latch "ALE + Read"). `addr` itself is preserved for the hd-pack
4728        // tile-base latch below.
4729        let read_addr = self.ale_splice(addr);
4730        self.bg_lo_latch = self.read_vram(bus, read_addr);
4731        // v2.3.2 "Lucid" — the pattern ROW address (fine-Y kept, unlike the
4732        // `hd-pack` latch below which masks it off to get the 16-byte tile base):
4733        // provenance answers "which CHR byte fed THIS pixel", which is a row, not
4734        // a tile. The SPLICED address again, so a hybrid-address corruption shows
4735        // what was really read.
4736        #[cfg(feature = "debug-hooks")]
4737        {
4738            self.prov_bg_latch = ProvBgAddrs {
4739                nt: self.prov_nt_pending,
4740                at: self.prov_at_pending,
4741                pattern: read_addr,
4742            };
4743        }
4744        self.ale_drive_data(self.bg_lo_latch);
4745        // v1.2.0 C3 (hd-pack): latch the 16-byte tile base (fine-Y masked off)
4746        // for this fetch group. Promoted into the `hd_bg_addr_*` queue at the
4747        // next shifter reload so it tracks the BG pattern shifters tile-for-tile.
4748        // Output-only; no new VRAM read, no A12.
4749        #[cfg(feature = "hd-pack")]
4750        {
4751            self.hd_bg_addr_latch = addr & 0x1FF0;
4752            // CHR-ROM absolute tile index (offset/16), or the CHR-RAM sentinel.
4753            self.hd_bg_idx_latch = bus.chr_phys(addr).map_or(HD_CHR_RAM, |o| o / 16);
4754        }
4755    }
4756
4757    /// Fetch BG pattern high byte (offset +8 from the low fetch).
4758    #[inline]
4759    fn fetch_bg_hi<B: PpuBus>(&mut self, bus: &mut B) {
4760        let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4761        let fine_y = self
4762            .bg_split_latch
4763            .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4764        let addr = bg_table | (u16::from(self.nt_latch) << 4) | 0x08 | fine_y;
4765        self.observe_a12_addr(bus, addr);
4766        // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half (see `fetch_bg_lo`):
4767        // A12 above stays on the INTENDED `addr`; only the DATA read address goes
4768        // through the ALE splice (stale-latch "ALE + Read").
4769        let read_addr = self.ale_splice(addr);
4770        self.bg_hi_latch = self.read_vram(bus, read_addr);
4771        self.ale_drive_data(self.bg_hi_latch);
4772    }
4773
4774    // === v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE fetch model ===============
4775    //
4776    // A genuine two-dot VRAM transaction. The attribute + pattern fetches' EVEN
4777    // dot (the ALE half, phases 2/4/6) drives the full 14-bit address onto
4778    // `address_bus` and captures its low byte into `octal_latch` via
4779    // [`Self::drive_bus`]; the following ODD dot (the read half, phases 3/5/7 —
4780    // the existing `fetch_*`) resolves the effective address through
4781    // [`Self::ale_splice`] and drives the DATA byte back onto the low bus via
4782    // [`Self::ale_drive_data`]. For a coherent fetch the address the ALE drove
4783    // equals the address the read would compute (`v` is constant across the 8-dot
4784    // group's phases 0..=6; the coarse-X increment is at phase 7 AFTER the read),
4785    // so the splice returns the intended address and this is behavior-neutral.
4786    //
4787    // The NAMETABLE fetch is the exception: its MMC5 vertical-split query
4788    // (`bg_split_state`, whose `split_chr_bank_latch` side effect
4789    // `nametable_fetch`/`chr_offset` read) is mapper-observable and must fire at
4790    // the read dot (phase 1), so the NT octal-latch load co-locates with the read
4791    // via `ale_splice`'s not-armed path (there is no phase-0 NT ALE); the NT ALE
4792    // (`ale_drive_nt`) still drives the plain address so the latch naturally
4793    // carries the one-tile-ahead NT-low the "Hybrid Addresses" corruption needs.
4794
4795    /// ALE half of the nametable fetch (phase 0). Drives the PLAIN nametable
4796    /// address `0x2000 | (v & 0x0FFF)` and loads the octal latch with its low
4797    /// byte — a true two-dot ALE for the common (non-split) case, which is what
4798    /// lets `octal_latch` naturally carry the one-tile-ahead NT-low the "Hybrid
4799    /// Addresses" corruption needs. The MMC5-split query is deferred to the read
4800    /// dot (phase 1, `fetch_nt`); when it turns out split-active, `fetch_nt`
4801    /// disarms this ALE and reads the synthesized `split.nt_addr` co-located.
4802    const fn ale_drive_nt(&mut self) {
4803        let nt_addr = 0x2000 | (self.v & 0x0FFF);
4804        self.drive_bus(nt_addr, false);
4805    }
4806
4807    /// ALE half of the attribute fetch (phase 2). Uses the `bg_split_latch`
4808    /// already set by the nametable read at phase 1.
4809    const fn ale_drive_at(&mut self) {
4810        let at_addr = if let Some(split) = self.bg_split_latch {
4811            split.at_addr
4812        } else {
4813            let v = self.v;
4814            0x23C0 | (v & 0x0C00) | ((v >> 4) & 0x38) | ((v >> 2) & 0x07)
4815        };
4816        self.drive_bus(at_addr, false);
4817    }
4818
4819    /// ALE half of the BG pattern-low fetch (phase 4). Uses `nt_latch` (set at
4820    /// the phase-1 nametable read) and `v`'s fine-Y (or the split fine-Y).
4821    fn ale_drive_bg_lo(&mut self) {
4822        let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4823        let fine_y = self
4824            .bg_split_latch
4825            .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4826        let addr = bg_table | (u16::from(self.nt_latch) << 4) | fine_y;
4827        self.drive_bus(addr, true);
4828    }
4829
4830    /// ALE half of the BG pattern-high fetch (phase 6). Same as the low plane
4831    /// with bit 3 set (the +8 byte offset).
4832    fn ale_drive_bg_hi(&mut self) {
4833        let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4834        let fine_y = self
4835            .bg_split_latch
4836            .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4837        let addr = bg_table | (u16::from(self.nt_latch) << 4) | 0x08 | fine_y;
4838        self.drive_bus(addr, true);
4839    }
4840
4841    /// Drive a full 14-bit fetch address onto the multiplexed bus (the ALE
4842    /// half): `address_bus` takes the whole address and the 74LS373 octal latch
4843    /// captures A7-A0. Arms `ale_armed` for the matching read half.
4844    ///
4845    /// `is_pattern` marks the two BG-pattern ALEs (phases 4/6). While the "ALE +
4846    /// Read" freeze (`pattern_latch_stale`) is pending — a `$2007`-read ALE
4847    /// overlapped the fetch cadence and froze `octal_latch` on the read's DATA
4848    /// byte — the latch is NOT reloaded (the frozen DATA byte survives across any
4849    /// intervening ALE); the first pattern ALE afterwards consumes the flag, so
4850    /// its read splices `(PAR high 6):(stale DATA low 8)` = `$0FFF`.
4851    const fn drive_bus(&mut self, addr: u16, is_pattern: bool) {
4852        self.address_bus = addr;
4853        if self.pattern_latch_stale {
4854            if is_pattern {
4855                self.pattern_latch_stale = false;
4856            }
4857        } else {
4858            self.octal_latch = (addr & 0xFF) as u8;
4859        }
4860        self.ale_armed = true;
4861    }
4862
4863    /// Resolve a fetch's effective read address through the multiplexed bus (the
4864    /// read half). When a real ALE preceded this read (`ale_armed`), the address
4865    /// is the splice of the ALE-driven high 6 bits with the latched low 8:
4866    /// `(address_bus & 0x3F00) | octal_latch` — transparent for a coherent fetch
4867    /// (Phase 1), the divergence point for the Phase-3 corruptions. With NO
4868    /// preceding ALE (the dot-337-340 garbage nametable fetches), drive + latch
4869    /// `intended` in place so the read stays behavior-neutral.
4870    #[allow(clippy::missing_const_for_fn)] // u16::from is not yet const-stable
4871    fn ale_splice(&mut self, intended: u16) -> u16 {
4872        if self.ale_armed {
4873            self.ale_armed = false;
4874            // v2.3.6 — high 6 bits from `intended` (recomputed from the LIVE `v` at
4875            // the read dot), not from the ALE-time `address_bus` snapshot. Upstream
4876            // AccuracyCoin's commentary was rewritten to say the address bus is
4877            // driven EVERY ppu cycle and its upper 6 bits track `v`, so the hybrid
4878            // address is what a continuously-driven bus produces when `v` moves
4879            // between a fetch's ALE half and its read half. Behaviour-neutral for a
4880            // coherent fetch: `v` unchanged => `intended` == the driven address.
4881            let effective = (intended & 0x3F00) | u16::from(self.octal_latch);
4882            // Diagnostic: record any read whose spliced effective address diverges
4883            // from the intended one (the two corruptions) for the TriCNES per-dot
4884            // cross-diff. `push` self-filters to scanlines 2-5, so this is cheap.
4885            if effective != intended {
4886                octal_trace::push(
4887                    if intended >= 0x2000 {
4888                        octal_trace::K_HYBRID
4889                    } else {
4890                        octal_trace::K_STALE
4891                    },
4892                    self.frame,
4893                    self.scanline,
4894                    self.dot,
4895                    u32::from(effective),
4896                );
4897            }
4898            effective
4899        } else {
4900            self.address_bus = intended;
4901            self.octal_latch = (intended & 0xFF) as u8;
4902            intended
4903        }
4904    }
4905
4906    /// Data half of a VRAM access: drive the byte just read back onto the
4907    /// multiplexed bus's low 8 bits (AD7-0). `octal_latch` is NOT refreshed here
4908    /// (the 74LS373 latch holds the ADDRESS low from the ALE) — that retention is
4909    /// what a `$2007`-read ALE overlap exploits (the "ALE + Read" corruption). It
4910    /// is otherwise transparent: the next fetch's ALE overwrites `address_bus`
4911    /// wholesale.
4912    #[allow(clippy::missing_const_for_fn)] // u16::from is not yet const-stable
4913    fn ale_drive_data(&mut self, data: u8) {
4914        self.address_bus = (self.address_bus & 0xFF00) | u16::from(data);
4915    }
4916
4917    /// Shift the BG pattern and attribute shift registers by one bit.
4918    ///
4919    /// All four registers are 16-bit and advance in lockstep so the
4920    /// attribute palette tracks the same tile column as the pattern bits.
4921    const fn shift_bg(&mut self) {
4922        self.bg_shift_lo <<= 1;
4923        self.bg_shift_hi <<= 1;
4924        self.at_shift_lo <<= 1;
4925        self.at_shift_hi <<= 1;
4926        // v2.0 Phase 6 (mc-ppu-subpos): BG-shifter SERIAL-IN. Per nesdev "PPU
4927        // signals", the bit shifted into the pattern shifters from the right is
4928        // a constant 0 for the LOW plane and a constant 1 for the HIGH plane.
4929        // It is normally invisible: the dot%8==1 reload overwrites bits 0-7
4930        // every 8 shifts, so the injected '1' never reaches the output bits
4931        // 8-15 before being washed (=> framebuffer byte-identical for normal
4932        // rendering, oracle-safe). It surfaces ONLY when a precisely-timed
4933        // `$2001` render-toggle SKIPS a reload while shifting continues, drawing
4934        // opaque BG pixels on an all-translucent nametable (AccuracyCoin "BG
4935        // Serial In"). The attribute shifters have no serial-in (the test only
4936        // needs a non-transparent pattern bit, not a specific palette).
4937        {
4938            self.bg_shift_hi |= 1;
4939        }
4940    }
4941
4942    /// Pre-fetch (dots 328 / 336) byte shift: advance all four BG shift
4943    /// registers by 8 bits in lockstep, moving the just-reloaded tile
4944    /// data from bits 0-7 to bits 8-15 and clearing bits 0-7 for the next
4945    /// reload. This substitutes for the per-cycle `shift_bg` that does not
4946    /// run during the dots 321-336 pre-fetch region. The attribute
4947    /// registers MUST shift identically to the pattern registers here —
4948    /// omitting them was the 086ce4d left-edge palette regression.
4949    #[inline]
4950    const fn prefetch_shift_bg_regs(&mut self) {
4951        self.bg_shift_lo <<= 8;
4952        self.bg_shift_hi <<= 8;
4953        self.at_shift_lo <<= 8;
4954        self.at_shift_hi <<= 8;
4955        // v1.2.0 C3 (hd-pack): the `<<= 8` promotes the low (next) tile into the
4956        // high (displayed) byte — mirror the address queue. Telemetry only.
4957        #[cfg(feature = "hd-pack")]
4958        {
4959            self.hd_bg_addr_cur = self.hd_bg_addr_next;
4960            self.hd_bg_idx_cur = self.hd_bg_idx_next;
4961        }
4962        // v2.3.2 "Lucid": same promotion for the provenance cascade.
4963        //
4964        // A/B'd rather than assumed. For the VISIBLE region this is redundant —
4965        // every displayed tile passes through a `reload_bg_shift_regs` that
4966        // overwrites `cur` from `next` anyway, and the provenance test passes
4967        // identically with this removed. It is kept because this function is also
4968        // called on the rendering-DISABLE edge (dots 329-336) to complete a
4969        // frozen group's pending shift, and there pixels can be emitted from the
4970        // shifters before any further reload — so mirroring the shift is what
4971        // keeps the reported tile matching the one actually on screen.
4972        #[cfg(feature = "debug-hooks")]
4973        {
4974            self.prov_bg_cur = self.prov_bg_next;
4975        }
4976    }
4977
4978    /// Reload the low bytes of the BG pattern and attribute shift
4979    /// registers from the latched fetch bytes.
4980    ///
4981    /// The 2-bit attribute is constant across all 8 pixels of a tile, so
4982    /// each attribute bit is expanded to a full `0xFF`/`0x00` byte into
4983    /// bits 0-7 — the same low-byte slot the pattern bytes occupy. This
4984    /// keeps the attribute shifter bit-for-bit aligned with the pattern
4985    /// shifters through both the per-cycle shifts (dots 1-256) and the
4986    /// pre-fetch `<<= 8` (dots 328 / 336).
4987    #[inline]
4988    const fn reload_bg_shift_regs(&mut self) {
4989        self.bg_shift_lo = (self.bg_shift_lo & 0xFF00) | self.bg_lo_latch as u16;
4990        self.bg_shift_hi = (self.bg_shift_hi & 0xFF00) | self.bg_hi_latch as u16;
4991        let at_lo = if (self.at_latch & 0x01) != 0 {
4992            0xFF
4993        } else {
4994            0x00
4995        };
4996        let at_hi = if (self.at_latch & 0x02) != 0 {
4997            0xFF
4998        } else {
4999            0x00
5000        };
5001        self.at_shift_lo = (self.at_shift_lo & 0xFF00) | at_lo;
5002        self.at_shift_hi = (self.at_shift_hi & 0xFF00) | at_hi;
5003        // v1.2.0 C3 (hd-pack): mirror the pattern reload — the prior `next`
5004        // tile is now in the high byte (displayed), and the freshly-latched
5005        // tile fills the low byte. Telemetry only; no state effect.
5006        #[cfg(feature = "hd-pack")]
5007        {
5008            self.hd_bg_addr_cur = self.hd_bg_addr_next;
5009            self.hd_bg_addr_next = self.hd_bg_addr_latch;
5010            self.hd_bg_idx_cur = self.hd_bg_idx_next;
5011            self.hd_bg_idx_next = self.hd_bg_idx_latch;
5012        }
5013        // v2.3.2 "Lucid": same promotion for the provenance address cascade —
5014        // one struct copy per stage, so the three addresses cannot drift apart.
5015        #[cfg(feature = "debug-hooks")]
5016        {
5017            self.prov_bg_cur = self.prov_bg_next;
5018            self.prov_bg_next = self.prov_bg_latch;
5019        }
5020    }
5021
5022    /// Increment coarse X with nametable-X wrap.
5023    ///
5024    /// Note: this is an internal loopy-register increment.  It does NOT
5025    /// drive the PPU address bus, so it must not emit A12 transitions —
5026    /// the address bus stays on the last-fetched address (BG-high) until
5027    /// the next fetch.  An earlier version of this code called
5028    /// `observe_a12` here, which spuriously interpreted `v`'s fine-Y bit
5029    /// 0 as A12 and produced ~16 false A12 rising edges per scanline,
5030    /// breaking MMC3's IRQ count (which expects exactly 1 rise per
5031    /// rendered scanline, at PPU dot ~260, with standard pattern-table
5032    /// layout).
5033    const fn inc_hori_v(&mut self) {
5034        if (self.v & 0x001F) == 31 {
5035            self.v &= !0x001F;
5036            self.v ^= 0x0400;
5037        } else {
5038            self.v += 1;
5039        }
5040    }
5041
5042    /// Increment fine Y, with the 29->0 wrap-and-flip-nametable-Y quirk.
5043    ///
5044    /// Same A12 caveat as [`Self::inc_hori_v`]: this is an internal
5045    /// register increment, not an address-bus driver.
5046    const fn inc_vert_v(&mut self) {
5047        if (self.v & 0x7000) == 0x7000 {
5048            self.v &= !0x7000;
5049            let mut y = (self.v & 0x03E0) >> 5;
5050            if y == 29 {
5051                y = 0;
5052                self.v ^= 0x0800;
5053            } else if y == 31 {
5054                y = 0;
5055            } else {
5056                y += 1;
5057            }
5058            self.v = (self.v & !0x03E0) | (y << 5);
5059        } else {
5060            self.v += 0x1000;
5061        }
5062    }
5063
5064    /// Copy horizontal bits of `t` into `v` (bits 0-4 + 10).
5065    const fn copy_hori_t_to_v(&mut self) {
5066        self.v = (self.v & !0x041F) | (self.t & 0x041F);
5067    }
5068
5069    /// Copy vertical bits of `t` into `v` (bits 5-9 + 11-14).
5070    const fn copy_vert_t_to_v(&mut self) {
5071        self.v = (self.v & !0x7BE0) | (self.t & 0x7BE0);
5072    }
5073
5074    // ------------------------------------------------------------------
5075    // Pixel emission.
5076    // ------------------------------------------------------------------
5077
5078    /// Emit one pixel into the framebuffer at the current `(scanline, dot)`.
5079    #[allow(clippy::cast_sign_loss)]
5080    #[allow(clippy::too_many_lines)] // + the ppu-sprite-shifter-counter X-counter/shift loop
5081    fn emit_pixel(&mut self) {
5082        let pixel_x = self.dot - 1;
5083        let pixel_y = self.scanline as u16; // already validated >= 0 by caller
5084        let fx = self.x;
5085        // BG pixel (bits 0-1 = pattern, bits 2-3 = palette)
5086        let (bg_idx, bg_pal) = if self.mask.contains(PpuMask::SHOW_BG)
5087            && (pixel_x >= 8 || self.mask.contains(PpuMask::SHOW_BG_LEFT))
5088        {
5089            let mask = 0x8000u16 >> fx;
5090            let p0 = u8::from((self.bg_shift_lo & mask) != 0);
5091            let p1 = u8::from((self.bg_shift_hi & mask) != 0);
5092            let idx = (p1 << 1) | p0;
5093            let a0 = u8::from((self.at_shift_lo & mask) != 0);
5094            let a1 = u8::from((self.at_shift_hi & mask) != 0);
5095            (idx, (a1 << 1) | a0)
5096        } else {
5097            (0, 0)
5098        };
5099
5100        // Sprite pixel evaluation (Sprint 2-3).
5101        let mut spr_idx: u8 = 0;
5102        let mut spr_pal: u8 = 0;
5103        let mut spr_priority_front = false;
5104        let mut spr_zero_pixel = false;
5105        #[cfg(any(feature = "hd-pack", feature = "debug-hooks"))]
5106        let mut spr_slot: usize = 0;
5107        // v1.8.9 — every opaque sprite covering this pixel (slot indices), for the
5108        // HD-pack multi-sprite conditions. Collected but never consulted by the
5109        // winner logic below, so the framebuffer stays byte-identical.
5110        #[cfg(feature = "hd-pack")]
5111        let mut hd_sprites: [usize; 4] = [0; 4];
5112        #[cfg(feature = "hd-pack")]
5113        let mut hd_spr_n: usize = 0;
5114        if self.mask.contains(PpuMask::SHOW_SPRITE)
5115            && (pixel_x >= 8 || self.mask.contains(PpuMask::SHOW_SPRITE_LEFT))
5116        {
5117            for i in 0..self.spr_count as usize {
5118                // v2.0 (ppu-sprite-shifter-counter): a sprite emits when its
5119                // X-counter is 0 OR it is in the persistent halted state. The
5120                // `spr_x == 0` term keeps the legacy px-0 emit timing (an X=0
5121                // sprite re-armed at dot 339 must emit at px 0, not px 1 — else a
5122                // spurious Sprite-0-Hit test-8 hit), while `spr_halted` carries
5123                // Stale Sprite t5/6. Default build: the legacy `spr_x == 0`.
5124                let emit_active = self.spr_x[i] == 0 || self.spr_halted[i];
5125                if !emit_active {
5126                    continue;
5127                }
5128                let lo = u8::from((self.spr_shift_lo[i] & 0x80) != 0);
5129                let hi = u8::from((self.spr_shift_hi[i] & 0x80) != 0);
5130                let val = (hi << 1) | lo;
5131                if val == 0 {
5132                    continue;
5133                }
5134                // hd-pack: record every opaque sprite covering this pixel.
5135                #[cfg(feature = "hd-pack")]
5136                if hd_spr_n < 4 {
5137                    hd_sprites[hd_spr_n] = i;
5138                    hd_spr_n += 1;
5139                }
5140                // The first opaque sprite (priority order) is the VISIBLE winner;
5141                // `spr_idx == 0` gates it so only the first sets the render state.
5142                if spr_idx == 0 {
5143                    spr_idx = val;
5144                    spr_pal = self.spr_attr[i] & 0x03;
5145                    spr_priority_front = (self.spr_attr[i] & 0x20) == 0;
5146                    #[cfg(any(feature = "hd-pack", feature = "debug-hooks"))]
5147                    {
5148                        spr_slot = i;
5149                    }
5150                    if i == 0 && self.spr_zero_in_line {
5151                        spr_zero_pixel = true;
5152                    }
5153                }
5154                // Default build stops at the winner (byte-identical); the hd-pack
5155                // build keeps scanning to collect the hidden sprites above (which
5156                // never touch the winner state, so the framebuffer is unchanged).
5157                #[cfg(not(feature = "hd-pack"))]
5158                break;
5159            }
5160        }
5161        // v3.1.0 (`T-SPRITE-LIMIT`): the sprites beyond the eighth, only where
5162        // none of the eight hardware sprites is opaque (they have the higher
5163        // OAM indexes, so the lower priority). Never sprite 0, so never a hit.
5164        // `spr_extra_count` is 0 unless the option is on.
5165        if spr_idx == 0
5166            && self.spr_extra_count != 0
5167            && self.mask.contains(PpuMask::SHOW_SPRITE)
5168            && (pixel_x >= 8 || self.mask.contains(PpuMask::SHOW_SPRITE_LEFT))
5169        {
5170            for e in 0..usize::from(self.spr_extra_count) {
5171                let off = pixel_x.wrapping_sub(u16::from(self.spr_extra_x[e]));
5172                if off >= 8 {
5173                    continue;
5174                }
5175                let bit = 7 - off;
5176                let lo = (self.spr_extra_lo[e] >> bit) & 1;
5177                let hi = (self.spr_extra_hi[e] >> bit) & 1;
5178                let val = (hi << 1) | lo;
5179                if val != 0 {
5180                    spr_idx = val;
5181                    spr_pal = self.spr_extra_attr[e] & 0x03;
5182                    spr_priority_front = (self.spr_extra_attr[e] & 0x20) == 0;
5183                    break;
5184                }
5185            }
5186        }
5187
5188        // Combine BG + sprite per priority.
5189        //
5190        // v2.3.2 "Lucid": the priority chain now yields the palette ADDRESS and
5191        // the single `read_palette` happens after it, instead of each arm reading
5192        // inline. Semantically identical, and it makes the address available to
5193        // the provenance record below — so the panel reports the exact `$3Fxx`
5194        // this pixel came from, including the `$3F10` family pre-mirroring and
5195        // the rendering-disabled backdrop-override address, rather than
5196        // re-deriving it from the priority result and getting the corners wrong.
5197        let pal_addr: u16 = if bg_idx == 0 && spr_idx == 0 {
5198            // Universal background ($3F00) — EXCEPT the palette backdrop-override
5199            // (F1.1): with rendering DISABLED and the VRAM address `v` pointing
5200            // into palette space ($3F00-$3FFF), the palette's shared address line
5201            // is driven by `v`, so hardware outputs the color at `v & 0x1F`
5202            // INSTEAD of the backdrop (`NESdev` "PPU palettes"; Mesen2 `NesPpu.cpp`
5203            // / ares output stage). This is a DISPLAY artifact only — palette RAM
5204            // is not mutated. It cannot fire while rendering is enabled: there
5205            // the fetch pipeline owns `v` and this branch means a transparent
5206            // pixel, which is the genuine backdrop. `read_palette` applies the
5207            // $10/$14/$18/$1C mirror + greyscale, so the override is mirror- and
5208            // greyscale-correct with no extra handling.
5209            if !self.mask.rendering_enabled() && (self.v & 0x3F00) == 0x3F00 {
5210                0x3F00 | (self.v & 0x1F)
5211            } else {
5212                0x3F00
5213            }
5214        } else if bg_idx == 0 {
5215            0x3F10 | (u16::from(spr_pal) << 2) | u16::from(spr_idx)
5216        } else if spr_idx == 0 {
5217            0x3F00 | (u16::from(bg_pal) << 2) | u16::from(bg_idx)
5218        } else {
5219            // Both opaque. Sprite-0 hit detection (constraints per nesdev).
5220            if spr_zero_pixel
5221                && pixel_x < 255
5222                && !(pixel_x < 8
5223                    && (!self.mask.contains(PpuMask::SHOW_BG_LEFT)
5224                        || !self.mask.contains(PpuMask::SHOW_SPRITE_LEFT)))
5225            {
5226                self.status.insert(PpuStatus::SPRITE_ZERO_HIT);
5227            }
5228            if spr_priority_front {
5229                0x3F10 | (u16::from(spr_pal) << 2) | u16::from(spr_idx)
5230            } else {
5231                0x3F00 | (u16::from(bg_pal) << 2) | u16::from(bg_idx)
5232            }
5233        };
5234        // One read at the end instead of one per arm. `read_palette` is a pure
5235        // read (the greyscale mask it consults is not touched by the sprite-0-hit
5236        // insert above), so hoisting it out of the branches is behaviour-
5237        // preserving as well as what clippy's `branches_sharing_code` wants.
5238        let final_idx = self.read_palette(pal_addr) & 0x3F;
5239
5240        // Write RGBA8 to framebuffer.
5241        let off = ((pixel_y as usize) * 256 + pixel_x as usize) * 4;
5242        // v2.8.0 Phase 4 — route through the precomputed
5243        // `(emphasis << 6) | color` lookup (built from the same pure
5244        // `palette_color_to_rgba`, so byte-identical to the old per-pixel
5245        // call for both the 2C02 composite default and the Vs./PC10 RGB
5246        // palettes) and store all four bytes with one bounds-checked slice
5247        // copy instead of four indexed stores.
5248        // v3.1.0 (`T-PAL-EMPHASIS`): the index is the PHYSICAL tint (bit 0
5249        // red, bit 1 green, bit 2 blue). PPUMASK bit 5 is red on the NTSC 2C02
5250        // and GREEN on the PAL 2C07 and the Dendy, bit 6 the reverse (NESdev
5251        // "Colour emphasis"), so those two exchange off NTSC. The region is
5252        // fixed per console, so the branch is constant.
5253        let raw = (self.mask.bits() >> 5) & 0x07;
5254        let emph = usize::from(if matches!(self.region, PpuRegion::Ntsc) {
5255            raw
5256        } else {
5257            (raw & 0b100) | ((raw & 0b001) << 1) | ((raw & 0b010) >> 1)
5258        });
5259        let lut_idx = (emph << 6) | usize::from(final_idx);
5260        let rgba = self.rgba_lut[lut_idx];
5261        self.framebuffer[off..off + 4].copy_from_slice(&rgba);
5262        // Parallel palette-index output for the `NES_NTSC` composite filter
5263        // (T-110-A1). Same `(emphasis << 6) | colour` value, in index space;
5264        // `off` is the RGBA byte offset, so `off >> 2` is the pixel index.
5265        // NOTE (v2.3.1 G4): making this store conditional on a consumer wanting
5266        // it was measured by deleting it outright — the ceiling any opt-in gate
5267        // could reach — and the ceiling is ZERO on the shipped configuration.
5268        // `perf` attributes ~0.78% to this line, but a line's sample share is not
5269        // its marginal cost: this is a sequential `u16` store the store buffer
5270        // absorbs off the critical path, so removing it frees nothing and the
5271        // samples simply redistribute. Not worth the correctness hazard of
5272        // gating a buffer the NTSC filter, the mobile API, `fast_dotloop_diff`
5273        // and a unit test all read. See `docs/performance.md`.
5274        self.index_framebuffer[off >> 2] = lut_idx as u16;
5275
5276        // v1.2.0 C3 (hd-pack): record the CHR tile that produced this pixel,
5277        // mirroring the BG-vs-sprite priority decision above. Output-only; this
5278        // reads only already-computed local state, so the framebuffer and all
5279        // timing are byte-identical whether the feature is on or off.
5280        #[cfg(feature = "hd-pack")]
5281        {
5282            // A pixel shows the SPRITE iff the sprite pixel is opaque AND
5283            // (the BG pixel is transparent OR the sprite has front priority) —
5284            // the same condition the `final_idx` priority match encodes.
5285            let shows_sprite = spr_idx != 0 && (bg_idx == 0 || spr_priority_front);
5286            // Multi-sprite telemetry: the identity of every opaque sprite covering
5287            // this pixel (front-to-back), for `spriteAtPosition` / `spriteNearby`.
5288            let hd_sprite_list: [HdSprite; 4] = {
5289                let mut arr = [HdSprite::default(); 4];
5290                for (k, slot) in hd_sprites.iter().take(hd_spr_n).enumerate() {
5291                    arr[k] = HdSprite {
5292                        chr_tile_index: self.hd_spr_idx[*slot],
5293                        palette_colors: self.hd_sprite_palette_colors(self.spr_attr[*slot] & 0x03),
5294                    };
5295                }
5296                arr
5297            };
5298            let hd_sprite_n = u8::try_from(hd_spr_n).unwrap_or(4);
5299            let rec = if shows_sprite {
5300                let attr = self.spr_attr[spr_slot];
5301                let flip_h = (attr & 0x40) != 0;
5302                // Column within the sprite (screen X minus the sprite's origin X),
5303                // then flip so the captured offset samples the UNFLIPPED
5304                // replacement directly (composite is flip-free).
5305                // v2.9.5: on the odd-frame-deferred pixel (scanline 0, X=0,
5306                // `spr_rearm_deferred` still set) every drawing sprite emits
5307                // its FIRST column there regardless of its X, so the column is
5308                // 0, not `pixel_x - x`. CodeRabbit on #575: with X=10 the old
5309                // arithmetic gave column 6 and sampled the wrong HD texel.
5310                let col = if self.spr_rearm_deferred {
5311                    0
5312                } else {
5313                    pixel_x.wrapping_sub(u16::from(self.hd_spr_x[spr_slot])) & 7
5314                };
5315                let off_x = if flip_h { 7 - col } else { col };
5316                HdTileSource {
5317                    chr_addr: self.hd_spr_addr[spr_slot],
5318                    palette: spr_pal,
5319                    is_sprite: true,
5320                    flip_h,
5321                    flip_v: (attr & 0x80) != 0,
5322                    palette_colors: self.hd_sprite_palette_colors(spr_pal),
5323                    offset_x: u8::try_from(off_x).unwrap_or(0),
5324                    offset_y: self.hd_spr_off_y[spr_slot], // already flip-baked at fetch
5325                    chr_tile_index: self.hd_spr_idx[spr_slot],
5326                    color_mask: self.mask.bits() & 0xE1,
5327                    sprites: hd_sprite_list,
5328                    sprite_count: hd_sprite_n,
5329                }
5330            } else if bg_idx != 0 {
5331                // Fine-X picks which of the two shifter tiles this pixel shows +
5332                // the column within it (Mesen usePrev / OffsetX); fine-Y is the row.
5333                let pos = u16::from(fx) + (pixel_x & 7);
5334                let chr = if pos < 8 {
5335                    self.hd_bg_addr_cur
5336                } else {
5337                    self.hd_bg_addr_next
5338                };
5339                HdTileSource {
5340                    chr_addr: chr,
5341                    palette: bg_pal,
5342                    is_sprite: false,
5343                    flip_h: false,
5344                    flip_v: false,
5345                    palette_colors: self.hd_bg_palette_colors(bg_pal),
5346                    offset_x: u8::try_from(pos & 7).unwrap_or(0),
5347                    offset_y: u8::try_from((self.v >> 12) & 7).unwrap_or(0),
5348                    chr_tile_index: if pos < 8 {
5349                        self.hd_bg_idx_cur
5350                    } else {
5351                        self.hd_bg_idx_next
5352                    },
5353                    color_mask: self.mask.bits() & 0xE1,
5354                    sprites: hd_sprite_list,
5355                    sprite_count: hd_sprite_n,
5356                }
5357            } else {
5358                // Universal background — no tile to substitute.
5359                HdTileSource {
5360                    chr_addr: HD_TILE_NONE,
5361                    palette: 0,
5362                    is_sprite: false,
5363                    flip_h: false,
5364                    flip_v: false,
5365                    offset_x: 0,
5366                    offset_y: 0,
5367                    chr_tile_index: HD_CHR_RAM,
5368                    palette_colors: 0,
5369                    color_mask: 0,
5370                    sprites: hd_sprite_list,
5371                    sprite_count: hd_sprite_n,
5372                }
5373            };
5374            self.hd_tile_source[off >> 2] = rec;
5375        }
5376
5377        // v2.3.2 "Lucid" phase 2 — the per-pixel causal record.
5378        //
5379        // Guarded on a plain `bool` rather than `prov_frame.is_some()`: this runs
5380        // 61,440 times a frame in one of the two hottest functions in the
5381        // emulator, so the unarmed cost is one predicted branch on an already-hot
5382        // cache line instead of an `Option` discriminant behind a pointer chase.
5383        // Everything recorded is already computed above or carried in the
5384        // fetch-time cascade — no new VRAM reads, no new arithmetic on the
5385        // shipped path — so the framebuffer and all timing are byte-identical
5386        // whether provenance is armed or not.
5387        #[cfg(feature = "debug-hooks")]
5388        if self.prov_armed {
5389            use crate::provenance::{PATTERN_ADDR_NONE, PixelLayer, PixelProvenance};
5390            // The same condition `final_idx` encoded above: a sprite is visible
5391            // iff it is opaque AND (the BG is transparent OR it has front
5392            // priority). Re-deriving it here rather than threading a flag keeps
5393            // the shipped path free of a variable that only telemetry reads.
5394            let shows_sprite = spr_idx != 0 && (bg_idx == 0 || spr_priority_front);
5395            let layer = if shows_sprite {
5396                PixelLayer::Sprite
5397            } else if bg_idx != 0 {
5398                PixelLayer::Background
5399            } else {
5400                PixelLayer::Backdrop
5401            };
5402            let rec = PixelProvenance {
5403                scanline: self.scanline,
5404                dot: self.dot,
5405                layer,
5406                palette_addr: pal_addr,
5407                palette_index: u8::try_from(palette_index(pal_addr)).unwrap_or(0),
5408                color: final_idx,
5409                color_mask: self.mask.bits() & 0xE1,
5410                // The DISPLAYED tile's addresses, from the cascade — `v` has
5411                // already advanced two tiles past this pixel.
5412                nt_addr: self.prov_bg_cur.nt,
5413                at_addr: self.prov_bg_cur.at,
5414                pattern_addr: match layer {
5415                    PixelLayer::Sprite => self.prov_spr_addr[spr_slot],
5416                    PixelLayer::Background => self.prov_bg_cur.pattern,
5417                    PixelLayer::Backdrop => PATTERN_ADDR_NONE,
5418                },
5419                bg_idx,
5420                bg_pal,
5421                spr_idx,
5422                spr_pal,
5423                sprite_slot: if shows_sprite {
5424                    u8::try_from(spr_slot).unwrap_or(0)
5425                } else {
5426                    crate::provenance::SPRITE_SLOT_NONE
5427                },
5428                sprite_front: spr_priority_front,
5429                sprite_zero: spr_zero_pixel,
5430                fine_x: fx,
5431                fine_y: u8::try_from((self.v >> 12) & 7).unwrap_or(0),
5432            };
5433            if let Some(frame) = self.prov_frame.as_mut() {
5434                frame.set(pixel_x as usize, pixel_y as usize, rec);
5435            }
5436        }
5437
5438        // Decrement sprite X-counters / shift sprite shift regs.
5439        //
5440        // v2.0 (ppu-sprite-shifter-counter): the X-COUNTER decrements every
5441        // visible dot regardless of rendering (Stale Sprite test 2 — forced
5442        // blank does NOT halt the counters), but the SHIFTER only advances while
5443        // rendering is ENABLED on the 1-PPU-dot-delayed gate (`rendering_enabled_
5444        // delayed`, the same gate `shift_bg` uses — test 3: the shifter PAUSES in
5445        // forced blank so a sprite's data survives a long blank and still draws
5446        // on re-enable). `spr_halted` is the persistent latch (set at counter==0
5447        // or across a disable; re-armed at dot 339) carrying Stale Sprite t5/6.
5448        // Default build: the legacy unconditional shift.
5449        for i in 0..self.spr_count as usize {
5450            if self.spr_halted[i] || self.spr_x[i] == 0 {
5451                // Halted / drawing: latch and shift while rendering is enabled.
5452                // The `spr_x == 0` term is load-bearing — a slot re-armed at dot
5453                // 339 with the counter already 0 must SHIFT this dot (not just
5454                // latch) to match the legacy `spr_x == 0 => shift` timing.
5455                self.spr_halted[i] = true;
5456                if self.rendering_enabled_delayed {
5457                    self.spr_shift_lo[i] <<= 1;
5458                    self.spr_shift_hi[i] <<= 1;
5459                }
5460            } else {
5461                // Counting: decrement every visible dot (forced blank does not
5462                // halt — test 2). On reaching 0, halt this tick.
5463                self.spr_x[i] -= 1;
5464                if self.spr_x[i] == 0 {
5465                    self.spr_halted[i] = true;
5466                }
5467            }
5468        }
5469        // v2.9.5: the re-arm that the odd-frame skip deferred takes effect now, after
5470        // pixel 0 was drawn and shifted in the drawing state. The counters then
5471        // start one dot late, which is what keeps pixels 1-7 in place.
5472        if self.spr_rearm_deferred {
5473            self.spr_rearm_deferred = false;
5474            for i in 0..self.spr_count as usize {
5475                self.spr_halted[i] = false;
5476            }
5477        }
5478    }
5479
5480    // ------------------------------------------------------------------
5481    // Sprite evaluation + tile fetch.
5482    // ------------------------------------------------------------------
5483
5484    /// Per-PPU-dot sprite-evaluation FSM.
5485    ///
5486    /// Reproduces the 2C02's three-phase sprite-eval state machine across
5487    /// dots 1..=256 of every visible scanline and the pre-render line:
5488    ///
5489    /// - **Dot 0**: reset FSM working state.
5490    /// - **Dots 1..=64**: clear secondary OAM to `$FF`. One byte cleared
5491    ///   every two dots (32 bytes over 64 dots). Reads of `$2004` during
5492    ///   this phase return `$FF` on real hardware.
5493    /// - **Dots 65..=256**: 192 dots = 96 read/write pairs. Odd dots read
5494    ///   a byte from primary OAM into a latch; even dots commit the latch
5495    ///   into secondary OAM (when copying is enabled). The buggy `n+m`
5496    ///   increment for overflow detection (when 8 sprites are already
5497    ///   latched) matches the documented hardware quirk that
5498    ///   `sprite_overflow_tests/4-Obscure` and `/5-Emulator` exercise.
5499    /// - **Dot 256**: commit `spr_count` and pre-clear unused slot
5500    ///   rendering-side arrays so the pixel pipeline never emits stale
5501    ///   sprite pixels.
5502    ///
5503    /// The actual per-slot pattern-table fetch (and its A12 transitions)
5504    /// happens later, in [`Self::fetch_sprite_tile`], unchanged. Sprite-
5505    /// tile fetches still dispatch at dots 260, 268, ..., 316 — preserving
5506    /// the canonical "241 A12 rises per NTSC frame" MMC3 IRQ count.
5507    /// v2.0 Tier 1.2 — value `$2004` returns while the screen is being drawn.
5508    ///
5509    /// Mirrors Mesen2 `NesPpu::ReadRam`'s `SpriteData` case
5510    /// (`NesPpu.cpp:361-380`): during the sprite-tile-load window (dots
5511    /// 257-320) the OAM data bus carries `secondary_oam[sprite*4 + min(step,3)]`
5512    /// (the 4th byte held for the 5 idle fetch cycles); at every other rendered
5513    /// dot it carries `oam_bus_copybuffer` (the sprite-eval data latch
5514    /// maintained by [`Self::tick_oam_bus`]). Caller has already checked
5515    /// `scanline <= 239 && rendering`.
5516    /// Is the isolated OAM-data-bus model the thing a `$2004` read observes
5517    /// right now?
5518    ///
5519    /// EXTRACTED so the register read and the diagnostic trace cannot drift.
5520    /// `tick_oam_bus` runs only on visible scanlines with rendering enabled, so
5521    /// off that window `oam_bus_copybuffer` holds whatever the last rendered dot
5522    /// left in it. `cpu_read_register` has always guarded against that; the
5523    /// v2.5.6 state-trace field did not, and would have reported a stale
5524    /// secondary-OAM byte as though it were the `$2004` value for the dot --
5525    /// which defeats the entire reason that field exists.
5526    #[must_use]
5527    pub(crate) const fn oam_data_bus_is_live(&self) -> bool {
5528        self.scanline <= 239 && self.is_render_scanline() && self.mask.rendering_enabled()
5529    }
5530
5531    /// What a `$2004` read would return at this exact dot, model included.
5532    ///
5533    /// This is the diagnostic counterpart of the `$2004` arm in
5534    /// `cpu_read_register`, minus that arm's side effects (open-bus touch, OAM
5535    /// decay refresh) -- a trace must not perturb what it observes.
5536    /// Feature-gated rather than `#[allow(dead_code)]`: its only caller is
5537    /// `build_state_record`, which is itself behind `ppu-state-trace`. An item
5538    /// that is live under one feature and dead by default is the case that
5539    /// earns an attribute — and compiling it out entirely is better than
5540    /// suppressing the warning about it.
5541    #[cfg(feature = "ppu-state-trace")]
5542    #[must_use]
5543    pub(crate) fn oam_data_bus_observed(&self) -> u8 {
5544        if self.oam_data_bus_is_live() {
5545            self.oam_data_bus_read()
5546        } else {
5547            let v = self.oam[self.oam_addr as usize];
5548            if (self.oam_addr & 0x03) == 0x02 {
5549                v & 0xE3
5550            } else {
5551                v
5552            }
5553        }
5554    }
5555
5556    fn oam_data_bus_read(&self) -> u8 {
5557        if (257..=320).contains(&self.dot) {
5558            let phase = (self.dot - 257) % 8;
5559            let step = if phase > 3 { 3 } else { phase };
5560            let oam_addr = ((self.dot - 257) / 8) * 4 + step;
5561            self.oam_bus_secondary[(oam_addr & 0x1F) as usize]
5562        } else {
5563            self.oam_bus_copybuffer
5564        }
5565    }
5566
5567    /// v2.0 Tier 1.2 — per-dot driver for the isolated OAM-data-bus model.
5568    ///
5569    /// A side-effect-free model of the `NESdev`-documented PPU sprite-evaluation
5570    /// sequence (`NESdev` wiki "PPU sprite evaluation" + "PPU rendering"):
5571    /// secondary-OAM clear (dots 1-64), evaluation (65-256), and sprite fetch
5572    /// (257-320) in the default configuration (the optional OAMADDR sprite-eval
5573    /// corruption glitch disabled; the 8-sprite overflow bug is still modeled),
5574    /// plus the cycle-321 copy-buffer reset. It maintains ONLY
5575    /// `oam_bus_copybuffer` +
5576    /// the parallel `oam_bus_secondary`; it reads primary `oam` read-only and
5577    /// NEVER touches the real sprite-eval / overflow / sprite-zero state (so
5578    /// the existing rendering FSM is unperturbed — `$2004` reads are the sole
5579    /// observable effect of this whole feature). Called each dot on visible
5580    /// scanlines (0-239) when rendering is enabled.
5581    /// Maintain `OAM2Address` and the "OAM2 Overflowed" freeze flag.
5582    ///
5583    /// Called from `tick_oam_bus`, so it inherits that call site's gate:
5584    /// visible scanline AND rendering enabled. That gate IS the ROM's
5585    /// "cleared if rendering is enabled during dots 63, 255, and 339"
5586    /// condition, which is why no explicit rendering test appears here --
5587    /// and why a scanline with rendering off correctly PRESERVES both the
5588    /// counter and the flag, which is the whole behaviour under test.
5589    ///
5590    /// Runs before `tick_oam_bus`'s clear- and eval-window early-returns,
5591    /// because two of the three reset dots (63 and 255) fall inside them.
5592    fn tick_oam2_address(&mut self, cycle: u16) {
5593        if matches!(cycle, 63 | 255 | 339) {
5594            self.oam2_fetch_addr = 0;
5595            self.oam2_overflowed = false;
5596        }
5597        if cycle == 257 {
5598            self.oam2_fetch_frozen = self.oam2_overflowed;
5599        }
5600        if (256..=320).contains(&cycle) && (cycle & 1) == 0 && !self.oam2_overflowed {
5601            // 32 bytes across the fetch window = one step per two dots. The
5602            // wrap raises the flag, which suppresses the 33rd increment and
5603            // leaves the address resting at 0 -- see the field docs.
5604            self.oam2_fetch_addr = (self.oam2_fetch_addr + 1) & 0x1F;
5605            if self.oam2_fetch_addr == 0 {
5606                self.oam2_overflowed = true;
5607            }
5608        }
5609        if cycle == 321 {
5610            // After fetch the `$2004` bus rests wherever OAM2Address actually
5611            // STOPPED -- index 0 in the ordinary case, because a complete
5612            // fetch wraps it there. This used to be a hard `[0]`, which is why
5613            // it was right in general and wrong exactly when the counter had
5614            // not completed its wrap.
5615            // `& 0x1F` matches every other `oam_bus_secondary` index site and
5616            // makes the bound structural: the counter is a 5-bit register and
5617            // `tick_oam2_address` already keeps it in range, while a restore
5618            // rejects anything wider. This is the third layer, not the first.
5619            self.oam_bus_copybuffer =
5620                self.oam_bus_secondary[(self.oam2_fetch_addr & 0x1F) as usize];
5621        }
5622    }
5623
5624    fn tick_oam_bus(&mut self) {
5625        let cycle = self.dot;
5626        // v2.3.0 (perf) — take the dot-0 early-out BEFORE deriving the sprite
5627        // height and y-test reference; both were computed unconditionally and
5628        // then discarded on this dot. Byte-identical: neither value is observable
5629        // on the path that returns here.
5630        if cycle == 0 {
5631            return;
5632        }
5633        self.tick_oam2_address(cycle);
5634        // NOTE (v2.3.1 G3): pushing these two below the `cycle < 65` early-out
5635        // as well — they are dead across the dots 1..=64 clear window — was
5636        // measured and produced NO change on any workload across two runs. LLVM
5637        // already sinks pure computations past branches that do not use them.
5638        // Do not re-attempt as a performance change; see `docs/performance.md`.
5639        let sprite_height: i16 = if self.ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
5640            16
5641        } else {
5642            8
5643        };
5644        // Y-test reference: the scanline being evaluated (sprites render on
5645        // scanline+1).
5646        let scan = self.scanline;
5647        if cycle < 65 {
5648            // Secondary-OAM clear (cycles 1-64): the bus carries $FF and the
5649            // parallel secondary OAM is filled with $FF, 1 byte per 2 dots.
5650            self.oam_bus_copybuffer = 0xFF;
5651            self.oam_bus_secondary[((cycle - 1) >> 1) as usize] = 0xFF;
5652            return;
5653        }
5654        if cycle <= 256 {
5655            if cycle & 1 == 1 {
5656                // Odd cycle: read a byte from primary OAM into the bus latch.
5657                if cycle == 65 {
5658                    // ProcessSpriteEvaluationStart: seed the eval pointer from
5659                    // OAMADDR (eval can begin mid-sprite if $2003 was written).
5660                    self.oam_bus_sprite_in_range = false;
5661                    self.oam_bus_secondary_addr = 0;
5662                    self.oam_bus_overflow_counter = 0;
5663                    self.oam_bus_copy_done = false;
5664                    self.oam_bus_addr_h = (self.oam_addr >> 2) & 0x3F;
5665                    self.oam_bus_addr_l = self.oam_addr & 0x03;
5666                }
5667                let addr = ((self.oam_bus_addr_l & 0x03) | (self.oam_bus_addr_h << 2)) as usize;
5668                // v2.1.4 F2.3 — OAM-decay read hook (no-op at the default): a
5669                // sprite-evaluation primary-OAM read refreshes the row's DRAM
5670                // cells (this is what keeps OAM alive during normal rendering).
5671                self.oam_decay_on_read((addr & 0xFF) as u8);
5672                let raw = self.oam[addr & 0xFF];
5673                // OAM byte 2 (attributes) bits 2-4 are unimplemented (read 0).
5674                self.oam_bus_copybuffer = if addr & 0x03 == 0x02 { raw & 0xE3 } else { raw };
5675            } else {
5676                // Even cycle: copy / decide.
5677                let cb = self.oam_bus_copybuffer as i16;
5678                let cb_in_range = scan >= cb && scan < cb + sprite_height;
5679                if self.oam_bus_copy_done {
5680                    self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5681                    // OAM write-disable turns secondary-OAM writes into reads.
5682                    // On early (pre-rev-G) 2C02s the data bus reads back the
5683                    // last byte the OAM-address counter rests on EVEN when fewer
5684                    // than 8 sprites were found (secondary_addr < 0x20) — the
5685                    // "OAM2[OAM2Address] every other cycle" behavior AccuracyCoin
5686                    // `$2004 Stress` section 6 documents. Mesen2 gates this on
5687                    // `secondary_addr >= 0x20` (rev-G+), which is why no Mesen
5688                    // config reproduces the section-6 `$03`; the test's answer
5689                    // key (the spec) wants the unconditional read. Each
5690                    // out-of-range sprite's Y was already written to
5691                    // `secondary[secondary_addr]` (the frozen index) below, so
5692                    // this reads back that last-written Y.
5693                    self.oam_bus_copybuffer =
5694                        self.oam_bus_secondary[(self.oam_bus_secondary_addr & 0x1F) as usize];
5695                } else {
5696                    if !self.oam_bus_sprite_in_range && cb_in_range {
5697                        self.oam_bus_sprite_in_range = true;
5698                    }
5699                    if self.oam_bus_secondary_addr < 0x20 {
5700                        // Copy one byte to (parallel) secondary OAM.
5701                        self.oam_bus_secondary[self.oam_bus_secondary_addr as usize] =
5702                            self.oam_bus_copybuffer;
5703                        if self.oam_bus_sprite_in_range {
5704                            self.oam_bus_addr_l += 1;
5705                            self.oam_bus_secondary_addr += 1;
5706                            // OAM2 full: the address overflowed, which raises
5707                            // the freeze flag. The dot-255 reset normally
5708                            // clears it again before sprite fetch; it survives
5709                            // only if rendering is disabled across that dot.
5710                            self.oam2_overflowed |= self.oam_bus_secondary_addr == 0x20;
5711                            if self.oam_bus_addr_l >= 4 {
5712                                self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5713                                self.oam_bus_addr_l = 0;
5714                                if self.oam_bus_addr_h == 0 {
5715                                    self.oam_bus_copy_done = true;
5716                                }
5717                            }
5718                            if self.oam_bus_secondary_addr.trailing_zeros() >= 2 {
5719                                // Finished copying all 4 bytes of this sprite.
5720                                self.oam_bus_sprite_in_range = false;
5721                                if self.oam_bus_addr_l != 0 && !cb_in_range {
5722                                    self.oam_bus_addr_l = 0;
5723                                }
5724                            }
5725                        } else {
5726                            // Nothing to copy — skip to the next sprite.
5727                            self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5728                            self.oam_bus_addr_l = 0;
5729                            if self.oam_bus_addr_h == 0 {
5730                                self.oam_bus_copy_done = true;
5731                            }
5732                        }
5733                    } else {
5734                        // 8 sprites found: secondary-OAM writes become reads.
5735                        self.oam_bus_copybuffer =
5736                            self.oam_bus_secondary[(self.oam_bus_secondary_addr & 0x1F) as usize];
5737                        if self.oam_bus_sprite_in_range {
5738                            // Overflow detected. (NOTE: the REAL SpriteOverflow
5739                            // flag is owned by the existing eval FSM — this
5740                            // isolated model deliberately does not set it.)
5741                            self.oam_bus_addr_l += 1;
5742                            if self.oam_bus_addr_l == 4 {
5743                                self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5744                                self.oam_bus_addr_l = 0;
5745                            }
5746                            if self.oam_bus_overflow_counter == 0 {
5747                                self.oam_bus_overflow_counter = 3;
5748                            } else {
5749                                self.oam_bus_overflow_counter -= 1;
5750                                if self.oam_bus_overflow_counter == 0 {
5751                                    self.oam_bus_copy_done = true;
5752                                    self.oam_bus_addr_l = 0;
5753                                }
5754                            }
5755                        } else {
5756                            // Sprite-eval bug: increment BOTH H and L.
5757                            self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5758                            self.oam_bus_addr_l = (self.oam_bus_addr_l + 1) & 0x03;
5759                            if self.oam_bus_addr_h == 0 {
5760                                self.oam_bus_copy_done = true;
5761                            }
5762                        }
5763                    }
5764                }
5765            }
5766        }
5767    }
5768
5769    // v2.3.0 (perf) — called once per ELIGIBLE dot on the fast dot path (visible
5770    // dots 1..=256 with rendering enabled: up to 61,440/frame, not all 89,342 —
5771    // idle lines and rendering-disabled paths bypass it entirely);
5772    // `perf annotate` showed its own prologue/epilogue (`push`/`ret`) as the two
5773    // hottest instructions in the body, i.e. pure call overhead LLVM had declined
5774    // to remove. `inline` lets it be folded into the dot loop. Byte-identical (an
5775    // inlining hint changes no behavior); adopted only if it clears the >3% bar.
5776    #[inline]
5777    pub(crate) fn tick_sprite_eval_per_dot(&mut self) {
5778        // Y-test reference line for sprite evaluation. Per nesdev
5779        // "PPU OAM" (Byte 0): "The first scanline that the sprite is
5780        // rendered on is one greater than this value." Hardware
5781        // performs the y-test `(scanline - y) in [0, h-1]` using the
5782        // CURRENT scanline counter — the eval at scanline N produces
5783        // sprites that render on scanline N+1. So sprite Y=N renders
5784        // on scanlines N+1..=N+h.
5785        //
5786        // Pre-render (scanline 261) prepares for scanline 0, but
5787        // scanline 0 never displays sprites per nesdev. We model
5788        // this by using -1 as the y-test reference, which makes
5789        // `-1 - y < 0` for all OAM y values, so the y-test always
5790        // fails at pre-render and scanline 0 sees no sprites.
5791        //
5792        // That line, and the sprite height, are computed inside the
5793        // `65..=256` arm below, their only use. They are dead on 149 of the
5794        // 341 dots, and computing them up front cost real time.
5795        //
5796        // v2.3.1 measured this sink (G3) as "no change" and this comment
5797        // used to forbid re-trying it. That run used the pre-v2.9.1
5798        // `ab_check.sh`, which timed the same binary on both sides. v2.9.8
5799        // re-measured it with the fixed tool: -1.0% to -3.1% on all four
5800        // frame workloads, shipped `_fast` paths included, in two runs
5801        // (`docs/performance.md`, v2.9.8 campaign).
5802        match self.dot {
5803            0 => {
5804                // Start-of-scanline: reset FSM working state. We do NOT
5805                // touch the rendering-side `spr_*` arrays or
5806                // `spr_zero_in_line` here — they were committed at the
5807                // PREVIOUS scanline's dot 256 and are about to be read
5808                // by this scanline's sprite-pixel evaluator on dots
5809                // 1..=256.
5810                self.sprite_eval_n = 0;
5811                self.sprite_eval_m = 0;
5812                self.sprite_eval_found = 0;
5813                self.sprite_eval_sec_idx = 0;
5814                self.sprite_eval_copying = false;
5815                self.sprite_eval_done = false;
5816                self.sprite_eval_overflow_search = false;
5817                self.sprite_eval_read_latch = 0xFF;
5818                self.sprite_eval_zero_found = false;
5819                // Phase 3a: capture eval base from OAMADDR at the
5820                // dot-0 reset so the dots 65-256 active loop starts
5821                // walking from the captured `(start_n, start_m)`
5822                // position.  Mesen2 captures at cycle 65 (in
5823                // ProcessSpriteEvaluationStart); we capture at dot 0
5824                // because our FSM does the eval-base read BEFORE
5825                // dot 65 (the first read at dot 65 already needs
5826                // the offset).  This matters when the CPU writes
5827                // $2003 mid-vblank to set OAMADDR before the next
5828                // scanline's eval begins.
5829                {
5830                    self.sprite_eval_n = (self.oam_addr >> 2) & 0x3F;
5831                    self.sprite_eval_m = self.oam_addr & 0x03;
5832                }
5833                self.sprite_eval_first_iter = true;
5834            }
5835            1..=64 => {
5836                // Clear phase. Even-dot writes a $FF into secondary OAM
5837                // (1 byte per 2 dots, 32 bytes over 64 dots). Odd dots
5838                // are idle reads (driving $FF onto the bus).
5839                //
5840                // The pre-2026-05-17 implementation also reset the
5841                // rendering-side `spr_*` arrays + `spr_count` +
5842                // `spr_zero_in_line` here at dot 64. That was a B8a
5843                // regression: the rendering loop at line 1146..=1220
5844                // READS those arrays on dots 1..=256 of the CURRENT
5845                // scanline, so resetting them mid-scanline destroyed
5846                // sprites for dots 64..=256 (the right ~75% of every
5847                // scanline). The dot 256 End-of-eval fixup below is
5848                // the correct time to commit the NEXT scanline's
5849                // values; the dot 64 reset has been removed.
5850                if (self.dot & 1) == 0 {
5851                    let idx = ((self.dot - 1) >> 1) as usize;
5852                    if idx < self.secondary_oam.len() {
5853                        self.secondary_oam[idx] = 0xFF;
5854                    }
5855                }
5856            }
5857            65..=256 => {
5858                if self.dot == 65 {
5859                    // v3.1.0: OAMADDR AT TICK 65 sets where evaluation
5860                    // starts (nesdev "PPU registers" -> OAMADDR, "Values
5861                    // during rendering"), so re-seed `(n, m)` here. The dot-0
5862                    // capture above stays as the reset value, but a `$2003`
5863                    // write (or a rendering-time `$2004` bump) during the
5864                    // dots 1-64 clear must still move the start. Until
5865                    // v3.1.0 only the dot-0 value counted, which was
5866                    // invisible while every test wrote `$2003` before dot 0:
5867                    // AccuracyCoin f5f41dc2 moved its "Misaligned OAM
5868                    // behavior" write to dots 28-29 of scanline 0 (one
5869                    // `JSR`/`RTS` pair later) and test 3 failed, evaluating
5870                    // from the stale address. The OAM-bus model above has
5871                    // always seeded at cycle 65.
5872                    self.sprite_eval_n = (self.oam_addr >> 2) & 0x3F;
5873                    self.sprite_eval_m = self.oam_addr & 0x03;
5874                }
5875                if !self.sprite_eval_done {
5876                    let next_line: i16 = if self.scanline == self.region.prerender_line() {
5877                        -1
5878                    } else {
5879                        self.scanline
5880                    };
5881                    let sprite_height: i16 = if self.ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
5882                        16
5883                    } else {
5884                        8
5885                    };
5886                    self.tick_sprite_eval_active_dot(next_line, sprite_height);
5887                }
5888
5889                if self.dot == 256 {
5890                    // End-of-eval fixup: commit spr_count and the
5891                    // eval-side sprite-0 latch onto the rendering-side
5892                    // arrays. Pre-clear slots we did NOT fill so unused
5893                    // ones produce no output even though
5894                    // `fetch_sprite_tile` always runs all 8 slots.
5895                    self.spr_count = self.sprite_eval_found;
5896                    self.spr_zero_in_line = self.sprite_eval_zero_found;
5897                    for i in (self.spr_count as usize)..8 {
5898                        self.spr_shift_lo[i] = 0;
5899                        self.spr_shift_hi[i] = 0;
5900                        self.spr_attr[i] = 0;
5901                        self.spr_x[i] = 0xFF;
5902                    }
5903                }
5904            }
5905            _ => {
5906                // Dots 257..=340: eval is idle; sprite tile fetches happen
5907                // elsewhere (`fetch_sprite_tile`, scheduled at dots 260,
5908                // 268, ..., 316 from the tick() main path).
5909            }
5910        }
5911    }
5912
5913    /// Per-active-dot helper for the per-PPU-dot FSM. Drives the
5914    /// alternating read/write semantics of dots 65..=256 when eval has
5915    /// not yet exhausted primary OAM or set overflow.
5916    #[allow(clippy::too_many_lines)] // Phase 3a feature-gated branches expand the line count beyond the threshold; refactoring into sub-helpers would require sharing 5+ mutable fields by reference, hurting readability.
5917    fn tick_sprite_eval_active_dot(&mut self, next_line: i16, sprite_height: i16) {
5918        if (self.dot & 1) == 1 {
5919            // Odd dot: read.
5920            // Per nesdev wiki "PPU sprite evaluation": during dots 65-256,
5921            // the hardware updates OAMADDR to track the current eval read
5922            // position. A CPU $2004 read at this time sees the OAM byte
5923            // at that walking index. We surface the eval position into
5924            // `oam_addr` so that CPU reads of $2004 during sprite eval
5925            // observe the same behavior as real silicon. The dot-257-320
5926            // OAMADDR-reset added in `Ppu::tick` washes this back to 0
5927            // after eval, preserving the post-eval semantics that the
5928            // existing $4014 OAM DMA / blargg sprite_hit_tests rely on.
5929            // Phase 3a: under the eval-base-from-OAMADDR feature, the
5930            // y-test address ALWAYS uses `n*4 + m` so a misaligned
5931            // start (`oam_addr & 0x03 != 0` at dot 0) reads the
5932            // appropriate byte of the start sprite as the Y candidate
5933            // (Mesen2 `_spriteAddrL` model).  Under the legacy path,
5934            // `m` is reset to 0 between sprites and the y-test always
5935            // reads byte 0; the legacy special-case is preserved for
5936            // bit-exact compatibility.
5937            let addr = ((self.sprite_eval_n as usize) * 4) + (self.sprite_eval_m as usize);
5938            // v2.1.4 F2.3 — OAM-decay read hook (no-op at the default): the legacy
5939            // per-dot sprite-eval read path also refreshes the row it reads, so both
5940            // eval models keep OAM alive identically during rendering.
5941            self.oam_decay_on_read((addr & 0xFF) as u8);
5942            self.sprite_eval_read_latch = self.oam[addr & 0xFF];
5943            // Expose the current eval index via the OAMADDR register
5944            // (truncated to u8 via the `& 0xFF` mask). This is the
5945            // documented hardware behavior — see AccuracyCoin
5946            // `TEST_ArbitrarySpriteZero` sub-test 2's lengthy comment
5947            // explaining the eval / OAMADDR interaction.
5948            self.oam_addr = (addr & 0xFF) as u8;
5949        } else {
5950            // Even dot: write/decide.
5951            let latch = self.sprite_eval_read_latch;
5952            if self.sprite_eval_overflow_search {
5953                // Treat the read byte as a y-coord candidate.
5954                let row = next_line - (latch as i16);
5955                if row >= 0 && row < sprite_height {
5956                    self.status.insert(PpuStatus::SPRITE_OVERFLOW);
5957                    self.sprite_eval_done = true;
5958                } else {
5959                    // Buggy n+m increment: increment BOTH.
5960                    self.sprite_eval_m = (self.sprite_eval_m + 1) & 0x03;
5961                    if self.sprite_eval_n == 63 {
5962                        self.sprite_eval_done = true;
5963                    } else {
5964                        self.sprite_eval_n += 1;
5965                    }
5966                }
5967            } else if self.sprite_eval_copying {
5968                // Copy byte (m == 1, 2, 3) into secondary OAM.
5969                let sec_idx = self.sprite_eval_sec_idx as usize;
5970                if sec_idx < self.secondary_oam.len() {
5971                    self.secondary_oam[sec_idx] = latch;
5972                }
5973                self.sprite_eval_sec_idx += 1;
5974                self.sprite_eval_m += 1;
5975                // Phase 3a: under the eval-base feature, continue
5976                // copying until the secondary OAM is aligned to a
5977                // sprite boundary (sec_idx % 4 == 0) — Mesen2's model
5978                // (`_secondaryOamAddr & 0x03 == 0` check at line
5979                // 1062).  This handles misaligned start where 4
5980                // sequential reads from `(start_n*4+start_m)` span
5981                // sprite boundaries.  Under the legacy path,
5982                // `m == 4` is identical to "sec_idx & 3 == 0" because
5983                // copying always starts at m=1 (after y-test at m=0),
5984                // so they're equivalent in the legacy case.
5985                let copy_done = self.sprite_eval_sec_idx.trailing_zeros() >= 2;
5986                if self.sprite_eval_m == 4 {
5987                    self.sprite_eval_m = 0;
5988                    self.sprite_eval_n = (self.sprite_eval_n + 1) & 0x3F;
5989                }
5990                if copy_done {
5991                    // Finished this sprite. found was already
5992                    // incremented when the y-byte landed.
5993                    self.sprite_eval_copying = false;
5994                    // v3.1.0: the fourth byte copied is the X position, and
5995                    // the PPU range-tests it exactly as it tests Y. Out of
5996                    // range: OAMADDR += 1 then &= $FC, re-aligning. IN range:
5997                    // only += 1, so a misaligned start STAYS misaligned
5998                    // (AccuracyCoin "Misaligned OAM behavior" tests 4-7, the
5999                    // "+4* behavior ... Only +1 with the X Position" rule,
6000                    // stated in the ROM's comments). `m` already holds the
6001                    // += 1 (the `m == 4` wrap above covers the aligned
6002                    // case, where both rules agree), so only the
6003                    // out-of-range case clears it. Until v3.1.0 this cleared
6004                    // `m` unconditionally; the ROM's pre-f5f41dc2 fail path
6005                    // returned into the test body without popping its return
6006                    // address, which recorded that failure as a pass.
6007                    let x_row = next_line - (latch as i16);
6008                    if !(x_row >= 0 && x_row < sprite_height) {
6009                        self.sprite_eval_m = 0;
6010                    }
6011                    // Under feature: the m==4 wrap above already
6012                    // advanced n once.  Don't double-increment.
6013                    // Under legacy: m never wrapped, so n advances
6014                    // here for the first (and only) time.
6015                    {
6016                        // n was already advanced in the m==4 wrap
6017                        // block above; just check terminal conditions.
6018                        if self.sprite_eval_found == 8 {
6019                            self.sprite_eval_overflow_search = true;
6020                        }
6021                        if self.sprite_eval_n == 0 {
6022                            // n wrapped past 63 to 0 — done.
6023                            self.sprite_eval_done = true;
6024                        }
6025                    }
6026                }
6027            } else {
6028                // Y-test for sprite n.
6029                let row = next_line - (latch as i16);
6030                let in_range = row >= 0 && row < sprite_height;
6031                if in_range && self.sprite_eval_found < 8 {
6032                    // Write y into secondary OAM and start copying
6033                    // bytes 1..=3 over the next 3 even-dot writes.
6034                    let sec_idx = self.sprite_eval_sec_idx as usize;
6035                    if sec_idx < self.secondary_oam.len() {
6036                        self.secondary_oam[sec_idx] = latch;
6037                    }
6038                    self.sprite_eval_sec_idx += 1;
6039                    // Sprite-zero-hit eligibility: per nesdev wiki +
6040                    // Mesen2 (`NesPpu::ProcessSpriteEvaluation` line
6041                    // 1040-1044, "If the first Y coordinate we load
6042                    // is in range, set the sprite 0 flag — this
6043                    // happens even if this isn't actually the first
6044                    // sprite in OAM (i.e. because OAMADDR was not 0
6045                    // when evaluation started)"), the sprite at the
6046                    // eval-start position is sprite-zero IFF its Y
6047                    // is in range — NOT "first in-range sprite found".
6048                    // If the start sprite is out-of-range, no sprite
6049                    // on this scanline is sprite-zero.  Under Phase 3a,
6050                    // gate on `sprite_eval_first_iter` (the first y-test
6051                    // of the scanline); the legacy path keeps the
6052                    // canonical `n == 0` check.
6053                    let is_first_inrange = self.sprite_eval_first_iter;
6054                    if is_first_inrange {
6055                        self.sprite_eval_zero_flag_on();
6056                    }
6057                    self.sprite_eval_found += 1;
6058                    self.sprite_eval_copying = true;
6059                    // Phase 3a: increment from CURRENT m (handles
6060                    // misaligned start where eval began at m != 0).
6061                    // Legacy path resets to m=1 (canonical "skip Y,
6062                    // copy bytes 1..=3" pattern).
6063                    {
6064                        self.sprite_eval_m += 1;
6065                        if self.sprite_eval_m == 4 {
6066                            // The Y byte was the LAST byte of slot `n`
6067                            // (evaluation started at m = 3). OAMADDR steps
6068                            // on into slot n + 1 and the copy continues:
6069                            // the PPU copies four bytes whatever the
6070                            // alignment. Until v3.1.0 this ended the copy
6071                            // here, putting one byte in secondary OAM
6072                            // instead of four (AccuracyCoin "Misaligned OAM
6073                            // behavior" test 7, offset 3; masked like test
6074                            // 6 by the ROM's pre-f5f41dc2 fail path).
6075                            self.sprite_eval_m = 0;
6076                            if self.sprite_eval_n == 63 {
6077                                self.sprite_eval_done = true;
6078                            } else {
6079                                self.sprite_eval_n += 1;
6080                            }
6081                        }
6082                    }
6083                } else if in_range && self.sprite_eval_found == 8 {
6084                    // Defensive: 9th in-range sprite at the y-tested
6085                    // cell. In practice the `found == 8` transition
6086                    // happens at the end of copying sprite 7, which
6087                    // flips into `overflow_search` mode, so this branch
6088                    // is unreachable. Kept for safety.
6089                    self.status.insert(PpuStatus::SPRITE_OVERFLOW);
6090                    self.sprite_eval_done = true;
6091                } else {
6092                    // Not in range: advance to the next sprite, and REALIGN.
6093                    //
6094                    // AccuracyCoin's README gives the rule for the
6095                    // secondary-OAM-NOT-full case: "the OAM address is
6096                    // incremented by 4 and bitwise ANDed with $FC" -- so the
6097                    // byte index CLEARS rather than being carried. This core
6098                    // advanced `n` and left `m` at whatever misaligned value
6099                    // evaluation started from.
6100                    //
6101                    // Invisible whenever OAMADDR is a multiple of four (`m` is
6102                    // already 0 at every y-test), so only misaligned OAM can
6103                    // observe it -- measured at 114 occurrences in a full
6104                    // battery run, with no test's verdict depending on it.
6105                    // Pinned directly by
6106                    // `misaligned_oam_out_of_range_advance_follows_both_rules`.
6107                    //
6108                    // The FULL case is the other rule in the same entry --
6109                    // "only increment the OAM address by 5" -- and is already
6110                    // implemented as the buggy n+m increment in
6111                    // `sprite_eval_overflow_search` above.
6112                    self.sprite_eval_m = 0;
6113                    if self.sprite_eval_n == 63 {
6114                        self.sprite_eval_done = true;
6115                    } else {
6116                        self.sprite_eval_n += 1;
6117                    }
6118                }
6119                // Phase 3a: clear the "first-iteration" flag AFTER the
6120                // first y-test fires (regardless of in-range result).
6121                // Per Mesen2 `_cycle == 66` semantics — sprite-zero is
6122                // set only on the FIRST y-test that lands in range,
6123                // and only if it's the FIRST iteration overall.
6124                self.sprite_eval_first_iter = false;
6125            }
6126        }
6127    }
6128
6129    /// Helper: set the per-scanline sprite-zero-in-line flag from the
6130    /// FSM. Sets the EVAL-side latch (`sprite_eval_zero_found`); the
6131    /// rendering-side flag (`spr_zero_in_line`) is committed from this
6132    /// latch at dot 256.
6133    const fn sprite_eval_zero_flag_on(&mut self) {
6134        self.sprite_eval_zero_found = true;
6135    }
6136
6137    /// Arm the OAM-corruption disable edge on a `$2001` write — faithful
6138    /// port of `TriCNES`'s `$2001` write path (`Emulator.cs` lines
6139    /// 9684-9696 / 1740-1755). When rendering was ON before the write and
6140    /// the new mask turns BOTH BG + sprites OFF while on a render line (NOT
6141    /// in vblank), set the disable flags. `_instant` is the
6142    /// data-bus-immediate path (OAM eval observes the disable the same
6143    /// cycle); the non-instant flag is the regular 1-dot-delayed path. The
6144    /// disable edge is captured against the live eval pointer during the
6145    /// dots 1-64 window and committed on re-enable; the `!pending` guard
6146    /// stops a write from re-arming over an already-captured corruption.
6147    const fn arm_oam_corruption_disable(&mut self, was_rendering: bool) {
6148        if was_rendering
6149            && !self.mask.rendering_enabled()
6150            && !self.oam_corruption_pending
6151            && (self.scanline < self.region.vblank_start_line()
6152                || self.scanline == self.region.prerender_line())
6153        {
6154            self.oam_corruption_disabled = true;
6155            self.oam_corruption_disabled_instant = true;
6156        }
6157    }
6158
6159    /// OAM-corruption per-dot driver — faithful port of `TriCNES`'s
6160    /// `PPU_Render_SpriteEvaluation` corruption handling (`Emulator.cs`
6161    /// lines 2664-2762). Called every render-line dot (independent of the
6162    /// rendering gate, so the disable edge is observed even after the
6163    /// sprite-eval FSM stops). Three responsibilities, in `TriCNES` order:
6164    ///
6165    /// 1. **Commit on re-enable.** If rendering is currently enabled and a
6166    ///    corruption is pending, apply it (`TriCNES` applies on the first
6167    ///    rendered dot once `PPU_Mask_Show*_Instant` is set again). The
6168    ///    pre-render-line dot-0 hook in `tick` handles the
6169    ///    re-enable-during-vblank case.
6170    /// 2. **Maintain `OAM2Address` across the dots 1-64 clear window.**
6171    ///    Reset at dot 1; incremented once per even clear dot, masked to
6172    ///    0x1F — exactly as `TriCNES` drives `OAM2Address` during dots 1-64.
6173    /// 3. **Capture the index at the disable edge.** When the disable
6174    ///    flag (`oam_corruption_disabled` / `_instant`, armed by the
6175    ///    `$2001` write) is set during the dots 1-64 window on a NON
6176    ///    pre-render line, set `oam_corruption_pending` and capture
6177    ///    `oam_corruption_index = oam2_addr` (the live secondary-OAM write
6178    ///    pointer). The pre-render line is excluded from the capture (it is
6179    ///    a read-only eval line for OAM-corruption purposes in `TriCNES`).
6180    fn tick_oam_corruption(&mut self, rendering: bool) {
6181        let pre_render = self.scanline == self.region.prerender_line();
6182
6183        // (1) Commit pending corruption once rendering is (re-)enabled
6184        // during the eval window. The pre-render dot-0 path in `tick`
6185        // covers the re-enable-in-VBlank case separately.
6186        if rendering && self.oam_corruption_pending && !pre_render && self.dot >= 1 {
6187            self.process_oam_corruption();
6188        }
6189
6190        // (2) + (3) only matter inside the dots 1-64 secondary-OAM clear
6191        // window. Outside it the disable flags simply persist until the
6192        // next eval window (or are committed above on re-enable).
6193        if (1..=64).contains(&self.dot) {
6194            if self.dot == 1 {
6195                // TriCNES resets OAM2Address at dot 1 of the eval window.
6196                self.oam2_addr = 0;
6197            }
6198
6199            // Capture the disable edge against the LIVE pointer, on a
6200            // non-pre-render line only (TriCNES: capture is skipped on the
6201            // read-only pre-render eval line).
6202            if (self.oam_corruption_disabled || self.oam_corruption_disabled_instant)
6203                && !pre_render
6204                && !self.oam_corruption_pending
6205            {
6206                self.oam_corruption_pending = true;
6207                self.oam_corruption_index = self.oam2_addr;
6208            }
6209            // The disable arming is single-shot: clear it once observed in
6210            // the eval window (TriCNES clears both flags when it fires).
6211            self.oam_corruption_disabled = false;
6212            self.oam_corruption_disabled_instant = false;
6213
6214            // Advance OAM2Address on even clear dots (mirrors TriCNES's
6215            // `OAM2[OAM2Address] = latch; OAM2Address = (OAM2Address+1) &
6216            // 0x1F` on even cycles of the dots 1-64 clear).
6217            if (self.dot & 1) == 0 {
6218                self.oam2_addr = (self.oam2_addr + 1) & 0x1F;
6219            }
6220        }
6221    }
6222
6223    /// Apply a pending OAM corruption — faithful port of `TriCNES`'s
6224    /// `CorruptOAM` (`Emulator.cs` lines 2635-2651): one OAM "row" of 8
6225    /// bytes is overwritten from row 0, plus the corresponding secondary-OAM
6226    /// byte. The index (`oam_corruption_index`) was captured at the disable
6227    /// edge from the live secondary-OAM eval pointer; index 0x20 wraps to 0.
6228    /// Clears the pending flag.
6229    fn process_oam_corruption(&mut self) {
6230        let mut index = self.oam_corruption_index as usize;
6231        if index == 0x20 {
6232            index = 0;
6233        }
6234        // OAM[index*8 + i] = OAM[i] for i in 0..8 (a no-op when index == 0,
6235        // matching TriCNES — it still runs, copying row 0 onto itself).
6236        let first_eight: [u8; 8] = [
6237            self.oam[0],
6238            self.oam[1],
6239            self.oam[2],
6240            self.oam[3],
6241            self.oam[4],
6242            self.oam[5],
6243            self.oam[6],
6244            self.oam[7],
6245        ];
6246        let dst = index * 8;
6247        if dst + 8 <= self.oam.len() {
6248            self.oam[dst..dst + 8].copy_from_slice(&first_eight);
6249        }
6250        // Also corrupt secondary OAM: OAM2[index] = OAM2[0].
6251        if index < self.secondary_oam.len() {
6252            self.secondary_oam[index] = self.secondary_oam[0];
6253        }
6254        self.oam_corruption_pending = false;
6255    }
6256
6257    /// Fetch one sprite slot's pattern bytes.  Always called for all 8
6258    /// slots — for unused slots the secondary-OAM bytes are $FF, producing
6259    /// a dummy fetch that still toggles A12 to the sprite pattern table on
6260    /// real hardware.  This is what generates the per-scanline A12 rising
6261    /// edge that MMC3's IRQ counter clocks on.
6262    #[allow(clippy::cast_sign_loss)]
6263    fn fetch_sprite_tile<B: PpuBus>(&mut self, bus: &mut B, slot: usize) {
6264        // Mirrors the y-test convention in `tick_sprite_eval_per_dot`:
6265        // `next_line` is the y-test reference = the CURRENT scanline
6266        // counter (or -1 for pre-render). The fetched row index is
6267        // `next_line - y`, which matches the row that will be
6268        // displayed on `next_line + 1` (the next scanline that
6269        // renders the eval result).
6270        //
6271        // v2.0 (ppu-sprite-shifter-counter): treat the pre-render line as
6272        // scanline `(prerender_line & 0xFF)` (NTSC 261 & 255 = 5) for the
6273        // sprite-tile in-range check, so a sprite whose pixel lands on row 5
6274        // loads into the shifters for scanline 0 (the stale secondary-OAM slots
6275        // filtered by the `load` gate below) — AccuracyCoin "Sprites On Scanline
6276        // 0". Default keeps the `-1` reference (scanline 0 sees no sprites).
6277        let next_line: i16 = if self.scanline == self.region.prerender_line() {
6278            self.region.prerender_line() & 0xFF
6279        } else {
6280            self.scanline
6281        };
6282        let sprite_height: i16 = if self.ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
6283            16
6284        } else {
6285            8
6286        };
6287        // With the "OAM2 Overflowed" flag latched, the OAM2 Address cannot
6288        // increment, so sprite fetch re-reads index 0 for EVERY object --
6289        // eight copies of OAM2[0] instead of eight distinct sprites. This is
6290        // the mechanism AccuracyCoin `Frozen OAM2 Increment` targets.
6291        //
6292        // The freeze path is verified end to end, by probe rather than by
6293        // argument: across a battery run the flag is raised 809 times and
6294        // reaches sprite fetch exactly once (the single construction test 2
6295        // builds), on scanline 196, with `spr_count` = 8, `spr_zero_in_line`
6296        // = true, `secondary_oam[0]` = $C1, and all eight slots loading
6297        // Y/tile/attr/X = $C1/$C1/$C1/$C1.
6298        //
6299        // Test 2 nevertheless still fails, and NOT for any sprite reason. Its
6300        // detector is a sprite-zero hit, which needs an opaque BACKGROUND
6301        // pixel under the sprite, and on scanline 197 the background is opaque
6302        // nowhere in x=190..205 (sprite pixels there: 51; background: 0).
6303        // `v` is one vertical increment ahead: fine-Y reads 3 where the test
6304        // needs 2, so the tile it placed for the hit sits a row off. The cause
6305        // is upstream of everything here -- a `$2001` rendering-ENABLE landing
6306        // on dot 256 takes effect one dot early, firing the dot-256 vertical
6307        // increment that hardware does not. The ROM says so in as many words:
6308        // "Rendering is enabled on dot 256, but the PPU's vertical scroll is
6309        // NOT incremented."
6310        //
6311        // That is a `$2001` write-timing gap, not a sprite-evaluation one, and
6312        // it is deliberately NOT patched at the dot-256 site: a compensating
6313        // edit there is the shape v2.5.7 recorded, where a wrong phase had
6314        // every window compensating for it. See docs/STATUS.md.
6315        // Unreachable in ordinary rendering: the flag is cleared at dots 63,
6316        // 255 and 339 whenever rendering is enabled, so reaching fetch with
6317        // it set requires rendering to be off across dot 255.
6318        // A frozen OAM2Address means every read during sprite fetch returns
6319        // index 0 -- the SAME byte four times per sprite, not the first
6320        // sprite's four bytes. AccuracyCoin states the end state explicitly:
6321        // an OAM2 of `C1 24 00 FF C0 C0 ...` is "processed as if it was
6322        // C1 C1 C1 C1 ..." for all 32 bytes. Reading `[0..3]` here would give
6323        // Y/tile/attr/X = C1/24/00/FF, which is a different sprite entirely.
6324        let (y_byte, tile, attr, xpos) = if self.oam2_fetch_frozen {
6325            let frozen = self.secondary_oam[0];
6326            (frozen, frozen, frozen, frozen)
6327        } else {
6328            let base = slot * 4;
6329            (
6330                self.secondary_oam[base],
6331                self.secondary_oam[base + 1],
6332                self.secondary_oam[base + 2],
6333                self.secondary_oam[base + 3],
6334            )
6335        };
6336        let y = y_byte as i16;
6337        let in_use = slot < self.spr_count as usize;
6338        let flip_v = (attr & 0x80) != 0;
6339        let flip_h = (attr & 0x40) != 0;
6340
6341        // For unused slots, the row delta isn't meaningful (Y=$FF makes it
6342        // negative or huge) — pin to 0 so the address arithmetic is well
6343        // defined.  The only thing that matters here is that the pattern
6344        // address lands in the sprite pattern table, which it does because
6345        // the sprite-table-select bit is set as PPUCTRL bit 3 (8x8 mode)
6346        // or tile bit 0 (8x16 mode); for the cleared $FF tile in 8x16 mode
6347        // bit 0 = 1 picks the $1000 table.
6348        let mut row: u16 = if in_use {
6349            (next_line.wrapping_sub(y)).clamp(0, sprite_height - 1) as u16
6350        } else {
6351            0
6352        };
6353
6354        let (table, tile_idx, in_tile_row) = if sprite_height == 16 {
6355            let table = u16::from(tile & 0x01) << 12;
6356            let mut tindex = tile & 0xFE;
6357            if flip_v && in_use {
6358                row = 15 - row;
6359            }
6360            if row >= 8 {
6361                tindex = tindex.wrapping_add(1);
6362                row -= 8;
6363            }
6364            (table, tindex, row)
6365        } else {
6366            let table = u16::from(self.ctrl.contains(PpuCtrl::SPRITE_PATTERN_HIGH)) << 12;
6367            let r = if flip_v && in_use { 7 - row } else { row };
6368            (table, tile, r)
6369        };
6370
6371        let addr_lo = table | (u16::from(tile_idx) << 4) | in_tile_row;
6372        let addr_hi = addr_lo | 0x08;
6373        self.observe_a12_addr(bus, addr_lo);
6374        // Sprite CHR fetch: route through `ppu_read_sprite` so MMC5
6375        // (and any other mapper with split sprite vs. BG CHR banking)
6376        // can use its sprite-specific bank registers.
6377        let mut lo = bus.ppu_read_sprite(addr_lo);
6378        self.observe_a12_addr(bus, addr_hi);
6379        let mut hi = bus.ppu_read_sprite(addr_hi);
6380        // W2 ($2007 Stress): stash the RAW (pre-h-flip) pattern bytes so the
6381        // per-dot sprite-fetch read cadence (`tick_sprite_fetch_read`) can
6382        // feed `render_data_bus` for the deferred `$2007` PPUDATA reload.
6383        {
6384            self.spr_fetch_lo_raw[slot] = lo;
6385            self.spr_fetch_hi_raw[slot] = hi;
6386        }
6387        // v2.0 (ppu-sprite-shifter-counter): gate the shifter load on the sprite
6388        // being in-range of `next_line`. On visible scanlines this is a no-op
6389        // (the eval already guarantees every `in_use` slot is in-range), but on
6390        // the pre-render line (feature on, `next_line = 5`) it filters the STALE
6391        // secondary-OAM slots so only sprites whose pixel lands on row 5 load for
6392        // scanline 0. Default (feature off): load every `in_use` slot.
6393        let load = in_use && {
6394            let r = next_line.wrapping_sub(y);
6395            r >= 0 && r < sprite_height
6396        };
6397        if load {
6398            if flip_h {
6399                lo = reverse_bits(lo);
6400                hi = reverse_bits(hi);
6401            }
6402            self.spr_shift_lo[slot] = lo;
6403            self.spr_shift_hi[slot] = hi;
6404            self.spr_attr[slot] = attr;
6405            self.spr_x[slot] = xpos;
6406            // v1.2.0 C3 (hd-pack): stash the 16-byte tile base (in-tile row
6407            // masked off) for this sprite slot so `emit_pixel` can name the
6408            // CHR tile. Telemetry only; no new VRAM read here.
6409            #[cfg(feature = "hd-pack")]
6410            {
6411                self.hd_spr_addr[slot] = addr_lo & 0x1FF0;
6412                self.hd_spr_x[slot] = xpos;
6413                // `in_tile_row` is the post-flip-V fetch row, i.e. exactly the
6414                // (unflipped) replacement texel row to sample.
6415                self.hd_spr_off_y[slot] = u8::try_from(in_tile_row & 0x07).unwrap_or(0);
6416                // CHR-ROM absolute tile index for the sprite, or the CHR-RAM
6417                // sentinel (the common mappers share BG/sprite CHR banking).
6418                self.hd_spr_idx[slot] = bus.chr_phys(addr_lo).map_or(HD_CHR_RAM, |o| o / 16);
6419            }
6420            // v2.3.2 "Lucid": the sprite's pattern ROW address (in-tile row
6421            // KEPT, unlike the `hd-pack` tile base above). Captured separately
6422            // so neither feature's telemetry depends on the other being on.
6423            #[cfg(feature = "debug-hooks")]
6424            {
6425                self.prov_spr_addr[slot] = addr_lo;
6426            }
6427        } else {
6428            #[cfg(feature = "hd-pack")]
6429            {
6430                self.hd_spr_addr[slot] = HD_TILE_NONE;
6431            }
6432            #[cfg(feature = "debug-hooks")]
6433            {
6434                self.prov_spr_addr[slot] = crate::provenance::PATTERN_ADDR_NONE;
6435            }
6436        }
6437        // Else: shift regs already cleared in tick_sprite_eval_per_dot.
6438
6439        // v3.1.0 (`T-SPRITE-LIMIT`): after the eighth REAL fetch, the extra
6440        // sprites for the same line. A no-op unless the option is on.
6441        if slot == 7 {
6442            self.fetch_extra_sprites(bus, next_line, sprite_height);
6443        }
6444    }
6445
6446    /// v3.1.0 (`T-SPRITE-LIMIT`, FE-02): collect and fetch the sprites beyond
6447    /// the eighth for the next scanline, for display only.
6448    ///
6449    /// Runs once per line, after the eighth real sprite fetch, and only when
6450    /// the option is on, evaluation found eight (so the hardware dropped
6451    /// some), the line is visible, and the board's CHR reads are pure
6452    /// ([`PpuBus::chr_reads_are_pure`]: MMC2 / MMC4 latch on CHR reads, the
6453    /// J.Y. ASIC clocks an IRQ on them, and two boards latch address bits, so
6454    /// on those the option draws eight as stock). The fetch calls neither
6455    /// `observe_a12_addr` nor anything else a mapper can see beyond the read,
6456    /// so A12, mapper IRQs and every emulated byte stay exactly stock.
6457    ///
6458    /// Which sprites: an aligned walk of primary OAM from entry 0, skipping
6459    /// the first eight in range (the ones the hardware draws when evaluation
6460    /// starts at OAMADDR 0, as it does on every normally rendered line). A line
6461    /// whose evaluation starts misaligned (a mid-frame `$2003` write, a test
6462    /// construction) can draw a slightly different set; it is a display
6463    /// enhancement, not hardware behaviour. The walk reads `oam` directly and
6464    /// deliberately bypasses the optional OAM-decay read hook, so drawing the
6465    /// extra sprites can never refresh a decaying DRAM row the game could
6466    /// later observe.
6467    fn fetch_extra_sprites<B: PpuBus>(&mut self, bus: &mut B, next_line: i16, height: i16) {
6468        self.spr_extra_count = 0;
6469        if !self.sprite_limit_disabled
6470            || self.spr_count < 8
6471            || !(0..240).contains(&self.scanline)
6472            || !bus.chr_reads_are_pure()
6473        {
6474            return;
6475        }
6476        let mut in_range = 0usize;
6477        for n in 0..64usize {
6478            let y = i16::from(self.oam[n * 4]);
6479            let row = next_line.wrapping_sub(y);
6480            if row < 0 || row >= height {
6481                continue;
6482            }
6483            in_range += 1;
6484            if in_range <= 8 {
6485                continue;
6486            }
6487            let count = usize::from(self.spr_extra_count);
6488            if count == MAX_EXTRA_SPRITES {
6489                break;
6490            }
6491            let tile = self.oam[n * 4 + 1];
6492            let attr = self.oam[n * 4 + 2] & 0xE3;
6493            let x = self.oam[n * 4 + 3];
6494            let flip_v = attr & 0x80 != 0;
6495            #[allow(clippy::cast_sign_loss)] // `row` is in 0..height, checked above
6496            let mut r = row as u16;
6497            let (table, tile_idx) = if height == 16 {
6498                if flip_v {
6499                    r = 15 - r;
6500                }
6501                let base = tile & 0xFE;
6502                let idx = if r >= 8 { base.wrapping_add(1) } else { base };
6503                r &= 7;
6504                (u16::from(tile & 0x01) << 12, idx)
6505            } else {
6506                if flip_v {
6507                    r = 7 - r;
6508                }
6509                (
6510                    u16::from(self.ctrl.contains(PpuCtrl::SPRITE_PATTERN_HIGH)) << 12,
6511                    tile,
6512                )
6513            };
6514            let addr = table | (u16::from(tile_idx) << 4) | r;
6515            let mut lo = bus.ppu_read_sprite(addr);
6516            let mut hi = bus.ppu_read_sprite(addr | 0x08);
6517            if attr & 0x40 != 0 {
6518                lo = reverse_bits(lo);
6519                hi = reverse_bits(hi);
6520            }
6521            self.spr_extra_lo[count] = lo;
6522            self.spr_extra_hi[count] = hi;
6523            self.spr_extra_attr[count] = attr;
6524            self.spr_extra_x[count] = x;
6525            self.spr_extra_count += 1;
6526        }
6527    }
6528
6529    fn advance_dot(&mut self) {
6530        // Count every PPU master cycle (one per dot processed) for the NES_NTSC
6531        // colour phase. Output-only / cosmetic; never gates emulation.
6532        self.dot_counter = self.dot_counter.wrapping_add(1);
6533
6534        // Odd-frame skip: when the frame is odd and rendering is enabled,
6535        // the pre-render scanline 261 dot 339 transitions to (0, 0)
6536        // immediately, skipping dot 340.
6537        //
6538        // The rendering check reads `mask_for_skip_check` (two-stage
6539        // pipeline of `mask`, shifted at the bottom of this function), not
6540        // `mask` directly. The two-PPU-clock visibility delay between a
6541        // `$2001` write and this check is what makes blargg
6542        // `ppu_vbl_nmi/10-even_odd_timing` pass: lockstep applies the
6543        // PPUMASK write at the *start* of a CPU cycle, while real hardware
6544        // latches at φ2 (end of cycle). Without the delay the dot-339 skip
6545        // detector observes the write up to two PPU clocks earlier than
6546        // hardware does, mispredicting the skip when the write straddles
6547        // dot 339.
6548        if self.scanline == self.region.prerender_line()
6549            && self.dot == 339
6550            && (self.frame & 1) == 1
6551            && self.skip_gate_mask().rendering_enabled()
6552            && self.region == PpuRegion::Ntsc
6553        {
6554            self.dot = 0;
6555            self.scanline = 0;
6556            self.frame = self.frame.wrapping_add(1);
6557            self.frame_complete = true;
6558            self.snapshot_ntsc_phase();
6559            self.dot0_replaced = true;
6560            // The skipped dot is where the loaded shifters would have seen the
6561            // dot-339 re-arm, so it has not happened yet: they enter scanline 0
6562            // drawing, and `emit_pixel` releases them after pixel 0 (see
6563            // `spr_rearm_deferred`). The skip requires rendering on, so the
6564            // 339 re-arm ran for these slots and this puts them back.
6565            if self.spr_count > 0 {
6566                for i in 0..self.spr_count as usize {
6567                    self.spr_halted[i] = true;
6568                }
6569                self.spr_rearm_deferred = true;
6570            }
6571            self.mask_for_skip_check = self.mask_skip_pipe1;
6572            self.mask_skip_pipe1 = self.mask;
6573            return;
6574        }
6575
6576        self.dot += 1;
6577        if self.dot > 340 {
6578            self.dot = 0;
6579            // Advance scanline.
6580            if self.scanline == self.region.prerender_line() {
6581                self.scanline = 0;
6582                self.frame = self.frame.wrapping_add(1);
6583                self.frame_complete = true;
6584                self.snapshot_ntsc_phase();
6585            } else if self.extra_scanlines != 0 && self.scanline + 1 == self.region.prerender_line()
6586            {
6587                // v1.7.0 F3 — PPU extra-scanlines overclock. The line just
6588                // before pre-render is a pure idle vblank line (not visible,
6589                // not the VBL-set line, not pre-render): repeating it emits no
6590                // pixels, sets/clears no flags, and fires no VBL/NMI/A12 event
6591                // — it only adds CPU run-time (the surrounding scheduler still
6592                // clocks the CPU every third dot). When the counter is exhausted
6593                // we fall through to the pre-render line as usual. This whole
6594                // branch is unreachable while `extra_scanlines == 0`, so the
6595                // default build is byte-identical.
6596                if self.extra_lines_remaining == 0 {
6597                    self.extra_lines_remaining = self.extra_scanlines;
6598                }
6599                self.extra_lines_remaining -= 1;
6600                if self.extra_lines_remaining == 0 {
6601                    // Done inserting: advance to pre-render as normal.
6602                    self.scanline += 1;
6603                }
6604                // else: hold on this idle line and run it again.
6605            } else {
6606                self.scanline += 1;
6607            }
6608        }
6609        self.mask_for_skip_check = self.mask_skip_pipe1;
6610        self.mask_skip_pipe1 = self.mask;
6611    }
6612
6613    /// Whether the dot-256 vertical increment rides the shared `rendering_gate`,
6614    /// which is the shipped behaviour.
6615    ///
6616    /// `u8::MAX` means "follow the shared gate"; any other value gives the
6617    /// increment its own depth in the shared history.
6618    #[cfg(feature = "phi2-write-sweep")]
6619    fn scroll_gate_follows_render_gate() -> bool {
6620        SCROLL_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) == u8::MAX
6621    }
6622
6623    /// Shipped build: always the shared gate, folded away.
6624    #[cfg(not(feature = "phi2-write-sweep"))]
6625    const fn scroll_gate_follows_render_gate() -> bool {
6626        true
6627    }
6628
6629    /// Whether the dot-339 sprite-counter re-arm rides the shared
6630    /// `rendering_gate`, which is the shipped behaviour.
6631    ///
6632    /// `SPRITE_REARM_LAG` uses `u8::MAX` for "follow the shared gate" rather
6633    /// than a depth, because the shipped wiring is not any depth of the history
6634    /// -- it is whatever `RENDER_GATE_LAG` resolved to this dot.
6635    #[cfg(feature = "phi2-write-sweep")]
6636    fn sprite_rearm_follows_render_gate() -> bool {
6637        SPRITE_REARM_LAG.load(core::sync::atomic::Ordering::Relaxed) == u8::MAX
6638    }
6639
6640    /// Shipped build: always the shared gate, folded away.
6641    #[cfg(not(feature = "phi2-write-sweep"))]
6642    const fn sprite_rearm_follows_render_gate() -> bool {
6643        true
6644    }
6645
6646    /// The mask the OAM2 counter's gate consults, at the swept depth. Depth 0
6647    /// is the live mask -- the shipped read -- so the default build is
6648    /// unchanged by construction.
6649    #[cfg(feature = "phi2-write-sweep")]
6650    fn oam2_gate_rendering(&self) -> bool {
6651        let depth = OAM2_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) as usize;
6652        if depth == 0 {
6653            self.mask.rendering_enabled()
6654        } else {
6655            self.sweep_mask_history[(depth - 1).min(self.sweep_mask_history.len() - 1)]
6656                .rendering_enabled()
6657        }
6658    }
6659
6660    /// Shipped build: the OAM2 counter carries NO gate of its own.
6661    ///
6662    /// v2.6.18. The call site sits inside the shared `render_line &&
6663    /// rendering_gate` block, so the one-dot-delayed gate is already applied
6664    /// and an additional conjunct could only make the counter's edges differ
6665    /// from every other consumer's. It used to conjoin the LIVE mask, which
6666    /// did exactly that on the DISABLE edge -- see the call site. Depth 2 of
6667    /// the swept history is this same value, which is why
6668    /// [`OAM2_GATE_LAG`]'s default is 2 rather than 0.
6669    // `&self` is unused BY DESIGN and cannot be dropped: the signature has to
6670    // match the `phi2-write-sweep` variant above, which reads the swept mask
6671    // history, so the call site stays identical across both builds. Making this
6672    // an associated function would fork the call site, which is how the two
6673    // builds start disagreeing about something other than the knob.
6674    #[allow(clippy::unused_self)]
6675    #[cfg(not(feature = "phi2-write-sweep"))]
6676    const fn oam2_gate_rendering(&self) -> bool {
6677        true
6678    }
6679
6680    /// The mask the odd-frame-skip gate consults, at the swept pipeline depth.
6681    /// Depth 2 returns `mask_for_skip_check` -- the shipped value -- so the
6682    /// default build is unchanged by construction rather than by claim.
6683    #[cfg(feature = "phi2-write-sweep")]
6684    fn skip_gate_mask(&self) -> PpuMask {
6685        match SKIP_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) {
6686            0 => self.mask,
6687            1 => self.mask_skip_pipe1,
6688            _ => self.mask_for_skip_check,
6689        }
6690    }
6691
6692    /// Shipped build: the two-stage value, inlined to the same load.
6693    #[cfg(not(feature = "phi2-write-sweep"))]
6694    const fn skip_gate_mask(&self) -> PpuMask {
6695        self.mask_for_skip_check
6696    }
6697
6698    const fn is_render_scanline(&self) -> bool {
6699        // Visible (0..=239) and pre-render line.
6700        self.scanline >= 0 && self.scanline <= self.region.last_visible_line()
6701            || self.scanline == self.region.prerender_line()
6702    }
6703}
6704
6705/// Resolve an address in `$3F00-$3FFF` to a palette RAM index, applying the
6706/// `$3F10/$14/$18/$1C → $3F00/$04/$08/$0C` mirror.
6707const fn palette_index(addr: u16) -> usize {
6708    let mut idx = (addr & 0x1F) as usize;
6709    if matches!(idx, 0x10 | 0x14 | 0x18 | 0x1C) {
6710        idx -= 0x10;
6711    }
6712    idx
6713}
6714
6715/// Reverse the bit order of a byte (used for horizontally-flipped sprites).
6716const fn reverse_bits(b: u8) -> u8 {
6717    b.reverse_bits()
6718}
6719
6720#[cfg(test)]
6721mod tests {
6722    use super::*;
6723
6724    // T-73-005 / T-73-006 (Phase 7): pin the per-region timing table so an
6725    // accidental edit to a region constant trips a test instead of silently
6726    // mis-timing PAL/Dendy. The runtime frame-structure consequences are
6727    // gated by the integration test in
6728    // `crates/rustynes-test-harness/tests/region_timing.rs`.
6729    #[test]
6730    fn ppu_region_constants_match_hardware() {
6731        // NTSC: 262 lines (pre-render 261), VBL@241, no odd-frame skip caveat.
6732        assert_eq!(PpuRegion::Ntsc.prerender_line(), 261);
6733        assert_eq!(PpuRegion::Ntsc.vblank_start_line(), 241);
6734        assert_eq!(PpuRegion::Ntsc.post_reset_mask_cycles(), 29_658);
6735        // PAL: 312 lines (pre-render 311), VBL@241, longer reset mask.
6736        assert_eq!(PpuRegion::Pal.prerender_line(), 311);
6737        assert_eq!(PpuRegion::Pal.vblank_start_line(), 241);
6738        assert_eq!(PpuRegion::Pal.post_reset_mask_cycles(), 33_132);
6739        // Dendy: 312 lines, but VBL starts at 291 (the distinguishing trait).
6740        assert_eq!(PpuRegion::Dendy.prerender_line(), 311);
6741        assert_eq!(PpuRegion::Dendy.vblank_start_line(), 291);
6742        assert_eq!(PpuRegion::Dendy.post_reset_mask_cycles(), 33_132);
6743        // Last visible line is 239 in every region.
6744        for r in [PpuRegion::Ntsc, PpuRegion::Pal, PpuRegion::Dendy] {
6745            assert_eq!(r.last_visible_line(), 239);
6746        }
6747    }
6748
6749    /// `AccuracyCoin`'s README states two rules for advancing `OAMADDR` when a
6750    /// sprite's Y is out of range during evaluation, and they differ by
6751    /// whether secondary OAM is already full:
6752    ///
6753    /// > "the OAM address is incremented by 4 and bitwise ANDed with `$FC`"
6754    ///
6755    /// > "If Secondary OAM is full ... you should instead only increment the
6756    /// > OAM address by 5."
6757    ///
6758    /// Both are invisible while `OAMADDR` is a multiple of four, because the
6759    /// byte index is already 0 at every y-test. They are only observable under
6760    /// MISALIGNED OAM, and measurement showed the whole corpus reaches that
6761    /// case just 114 times in a full `AccuracyCoin` run while no test's verdict
6762    /// depends on it — so this pins it directly instead.
6763    #[test]
6764    fn misaligned_oam_out_of_range_advance_follows_both_rules() {
6765        /// Drive one y-test from a misaligned `OAMADDR` against a Y that is
6766        /// out of range, and report the resulting `(n, m)`.
6767        fn out_of_range_advance(full: bool) -> (u8, u8) {
6768            let mut ppu = Ppu::new(PpuRegion::Ntsc);
6769            ppu.mask = PpuMask::SHOW_SPRITE;
6770            ppu.scanline = 10;
6771            // Misaligned start: OAMADDR $05 seeds n = 1, m = 1.
6772            ppu.oam_addr = 0x05;
6773            // The byte the y-test reads is OAM[n*4 + m] = OAM[5]. Make it far
6774            // out of range for scanline 10.
6775            ppu.oam[5] = 0xF0;
6776
6777            // Dot 0 resets the FSM and captures the eval base from OAMADDR.
6778            ppu.dot = 0;
6779            ppu.tick_sprite_eval_per_dot();
6780            assert_eq!(
6781                (ppu.sprite_eval_n, ppu.sprite_eval_m),
6782                (1, 1),
6783                "seeded misaligned"
6784            );
6785
6786            if full {
6787                // Secondary OAM full: evaluation is in overflow-search mode,
6788                // which is the state the "+5" rule describes.
6789                ppu.sprite_eval_found = 8;
6790                ppu.sprite_eval_overflow_search = true;
6791            }
6792
6793            // Odd dot reads the byte; the following even dot runs the y-test.
6794            ppu.dot = 65;
6795            ppu.tick_sprite_eval_per_dot();
6796            ppu.dot = 66;
6797            ppu.tick_sprite_eval_per_dot();
6798            (ppu.sprite_eval_n, ppu.sprite_eval_m)
6799        }
6800
6801        // NOT full: +4 then AND $FC -- n advances and the byte index CLEARS.
6802        assert_eq!(
6803            out_of_range_advance(false),
6804            (2, 0),
6805            "secondary OAM not full: OAMADDR += 4 & $FC, so the misaligned byte \
6806             index must be cleared, not carried"
6807        );
6808
6809        // FULL: +5 -- n AND m both advance (the classic overflow bug).
6810        assert_eq!(
6811            out_of_range_advance(true),
6812            (2, 2),
6813            "secondary OAM full: OAMADDR += 5, so both the sprite index and the \
6814             byte index advance"
6815        );
6816    }
6817
6818    /// v3.1.0 — the three misaligned-evaluation rules the `AccuracyCoin`
6819    /// `f5f41dc2` re-sync exposed, each pinned on its own so a regression
6820    /// names the rule it broke:
6821    ///
6822    /// 1. evaluation starts at OAMADDR **as of dot 65**, not dot 0 (nesdev
6823    ///    "PPU registers" -> OAMADDR), so a write during the clear counts;
6824    /// 2. an in-range Y copies **four** bytes whatever the alignment, also from
6825    ///    `m = 3`, where the copy crosses into slot `n + 1`;
6826    /// 3. the fourth byte (X) is range-tested: in range, OAMADDR only steps by
6827    ///    one and stays misaligned; out of range, it steps and ANDs with `$FC`.
6828    ///
6829    /// Each assertion failed against the pre-v3.1.0 FSM (dot-0 seed, a
6830    /// one-byte copy from `m = 3`, an unconditional realign).
6831    #[test]
6832    fn misaligned_oam_eval_starts_at_dot_65_copies_four_bytes_and_tests_x() {
6833        /// A PPU on scanline 10 whose dot-0 reset has already run with
6834        /// OAMADDR 0, so only a later seed can pick up `oam_addr`.
6835        fn ppu_after_dot0(oam_addr: u8, oam: &[(usize, u8)]) -> Ppu {
6836            let mut ppu = Ppu::new(PpuRegion::Ntsc);
6837            ppu.mask = PpuMask::SHOW_SPRITE;
6838            ppu.scanline = 10;
6839            ppu.oam.fill(0xFF);
6840            for &(i, v) in oam {
6841                ppu.oam[i] = v;
6842            }
6843            ppu.oam_addr = 0;
6844            ppu.dot = 0;
6845            ppu.tick_sprite_eval_per_dot();
6846            // The write lands during the clear, after the dot-0 reset.
6847            ppu.oam_addr = oam_addr;
6848            ppu
6849        }
6850        fn run_dots(ppu: &mut Ppu, from: u16, to: u16) {
6851            for d in from..=to {
6852                ppu.dot = d;
6853                ppu.tick_sprite_eval_per_dot();
6854            }
6855        }
6856
6857        // (1) Seed at dot 65: OAMADDR 2 written after dot 0. Y at OAM[2] is in
6858        // range for scanline 10 (Y = 8), and the walk must start there.
6859        let mut ppu = ppu_after_dot0(0x02, &[(2, 8), (3, 0x11), (4, 0x22), (5, 0x33)]);
6860        run_dots(&mut ppu, 65, 66);
6861        assert_eq!(
6862            ppu.secondary_oam[0], 8,
6863            "evaluation must read its first Y from OAMADDR as of dot 65 (OAM[2])"
6864        );
6865
6866        // (2) Four bytes from m = 3: OAM[3] is Y, OAM[4..=6] belong to slot 1.
6867        let mut ppu = ppu_after_dot0(0x03, &[(3, 8), (4, 0xA1), (5, 0xA2), (6, 0xA3)]);
6868        run_dots(&mut ppu, 65, 72);
6869        assert_eq!(
6870            ppu.secondary_oam[..4],
6871            [8, 0xA1, 0xA2, 0xA3],
6872            "a misaligned in-range sprite copies four bytes, across the slot edge"
6873        );
6874
6875        // (3) X range test, from OAMADDR 1: Y = OAM[1], X = OAM[4].
6876        let x_case = |x: u8| {
6877            let mut ppu = ppu_after_dot0(0x01, &[(1, 8), (2, 0x11), (3, 0x22), (4, x)]);
6878            run_dots(&mut ppu, 65, 72);
6879            u16::from(ppu.sprite_eval_n) * 4 + u16::from(ppu.sprite_eval_m)
6880        };
6881        assert_eq!(
6882            x_case(8),
6883            0x05,
6884            "X in range: OAMADDR += 1 only, so the walk stays misaligned at $05"
6885        );
6886        assert_eq!(
6887            x_case(0xF0),
6888            0x04,
6889            "X out of range: OAMADDR += 1 then & $FC, realigning to $04"
6890        );
6891    }
6892
6893    /// v2.6.18 — the depth-2 rendering-gate pipeline must SHIFT, not freeze.
6894    ///
6895    /// Drives the named pair directly rather than `tick`, so it needs no bus
6896    /// and — the point — never stores `RENDER_GATE_LAG`, which 13 other tests
6897    /// in this binary would observe concurrently.
6898    ///
6899    /// Against the pre-fix code (`render_gate_prev2 = self.
6900    /// rendering_enabled_delayed` at tick-end, read back AFTER the depth-2
6901    /// re-point had already overwritten it) the field is assigned to itself and
6902    /// the gate reads `false` for every dot of the run. That made every
6903    /// `lag >= 2` cell of the derivation sweep a measurement of a permanently
6904    /// disabled gate. The assertions below fail on the first `true`.
6905    #[cfg(feature = "phi2-write-sweep")]
6906    #[test]
6907    fn render_gate_lag_shifts_a_two_dot_pipeline() {
6908        let mut ppu = Ppu::new(PpuRegion::Ntsc);
6909        // Power-on: both stages clear, rendering off.
6910        assert!(!ppu.rendering_enabled_delayed);
6911        assert!(!ppu.render_gate_prev2);
6912
6913        // One dot of the pipeline at depth 2, returning the value this dot's
6914        // gate reads. Mirrors `tick`'s order exactly: re-point, then fill
6915        // stage N-1 from the live mask, then shift N-1 into N-2.
6916        let dot = |ppu: &mut Ppu, rendering: bool| -> bool {
6917            ppu.mask = if rendering {
6918                PpuMask::SHOW_BG
6919            } else {
6920                PpuMask::empty()
6921            };
6922            let prev1 = ppu.render_gate_begin_dot(2);
6923            let gate = ppu.rendering_enabled_delayed;
6924            // `tick` shifts BEFORE it refills stage N-1; keep that order, or
6925            // the mutation below stops reproducing the shipped defect.
6926            ppu.render_gate_end_dot(prev1);
6927            ppu.rendering_enabled_delayed = ppu.mask.rendering_enabled();
6928            gate
6929        };
6930
6931        // Rendering goes ON and stays on: the gate must follow two dots later.
6932        assert!(!dot(&mut ppu, true), "dot 1: gate still sees 2 dots ago");
6933        assert!(!dot(&mut ppu, true), "dot 2: gate still sees 2 dots ago");
6934        assert!(
6935            dot(&mut ppu, true),
6936            "dot 3: the ON edge has shifted through"
6937        );
6938        assert!(dot(&mut ppu, true), "dot 4: stays on");
6939
6940        // Rendering goes OFF: the same two-dot delay, in the other direction.
6941        assert!(dot(&mut ppu, false), "dot 5: gate still sees rendering on");
6942        assert!(dot(&mut ppu, false), "dot 6: gate still sees rendering on");
6943        assert!(
6944            !dot(&mut ppu, false),
6945            "dot 7: the OFF edge has shifted through"
6946        );
6947    }
6948
6949    #[test]
6950    fn reset_clears_the_rendering_gate_pipeline() {
6951        // `reset` preserves dot/scanline, so gate history that survives it is
6952        // history from before the reset being applied after it. Every stage
6953        // must come back false, or a reset mid-scanline can fire the dot-256
6954        // vertical increment with PPUMASK empty.
6955        let mut ppu = Ppu::new(PpuRegion::Ntsc);
6956        ppu.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
6957        ppu.prev_rendering_enabled = true;
6958        ppu.rendering_enabled_delayed = true;
6959        ppu.rendering_enabled_delayed2 = true;
6960
6961        ppu.reset();
6962
6963        assert!(ppu.mask.is_empty(), "reset clears PPUMASK");
6964        assert!(
6965            !ppu.prev_rendering_enabled,
6966            "reset must clear prev_rendering_enabled"
6967        );
6968        assert!(
6969            !ppu.rendering_enabled_delayed,
6970            "reset must clear rendering-gate stage 1"
6971        );
6972        assert!(
6973            !ppu.rendering_enabled_delayed2,
6974            "reset must clear rendering-gate stage 2 -- otherwise a reset at dot \
6975             254 reaches dot 256 with rendering history from before the reset"
6976        );
6977    }
6978
6979    #[test]
6980    fn odd_frame_dot_skip_is_ntsc_only() {
6981        // The pre-render dot-339 odd-frame skip only fires on NTSC with
6982        // rendering enabled. Drive a rendering-enabled odd pre-render frame in
6983        // each region and confirm only NTSC collapses dot 340.
6984        fn skips(region: PpuRegion) -> bool {
6985            let mut ppu = Ppu::new(region);
6986            // Force an odd frame, rendering on, parked at pre-render dot 339.
6987            ppu.frame = 1;
6988            ppu.mask = PpuMask::SHOW_BG;
6989            ppu.mask_for_skip_check = PpuMask::SHOW_BG;
6990            ppu.scanline = region.prerender_line();
6991            ppu.dot = 339;
6992            ppu.advance_dot();
6993            // A skip lands us at (scanline 0, dot 0); no skip steps to dot 340.
6994            ppu.scanline == 0 && ppu.dot == 0
6995        }
6996        assert!(skips(PpuRegion::Ntsc), "NTSC odd frame skips dot 340");
6997        assert!(!skips(PpuRegion::Pal), "PAL never skips");
6998        assert!(!skips(PpuRegion::Dendy), "Dendy never skips");
6999    }
7000
7001    /// Test bus that owns 8 KiB of CHR-RAM with horizontal mirroring map.
7002    /// CIRAM lives in the PPU; this bus only services CHR + A12.
7003    struct TestBus {
7004        chr: [u8; 0x2000],
7005        a12_count: u32,
7006        last_a12: bool,
7007    }
7008
7009    impl TestBus {
7010        fn new() -> Self {
7011            Self {
7012                chr: [0u8; 0x2000],
7013                a12_count: 0,
7014                last_a12: false,
7015            }
7016        }
7017    }
7018
7019    impl PpuBus for TestBus {
7020        fn ppu_read(&mut self, addr: u16) -> u8 {
7021            if addr < 0x2000 {
7022                self.chr[addr as usize]
7023            } else {
7024                0
7025            }
7026        }
7027        fn ppu_write(&mut self, addr: u16, value: u8) {
7028            if addr < 0x2000 {
7029                self.chr[addr as usize] = value;
7030            }
7031        }
7032        fn notify_a12(&mut self, level: bool) {
7033            if level != self.last_a12 {
7034                self.a12_count += 1;
7035                self.last_a12 = level;
7036            }
7037        }
7038        fn nametable_address(&self, addr: u16) -> u16 {
7039            // Horizontal mirroring: tables 0/1 -> bank 0, 2/3 -> bank 1.
7040            let table = ((addr.wrapping_sub(0x2000)) / 0x0400) & 0x03;
7041            let local = addr & 0x03FF;
7042            let phys = u16::from(table >= 2);
7043            phys * 0x0400 + local
7044        }
7045    }
7046
7047    fn fresh_ppu() -> (Ppu, TestBus) {
7048        let mut ppu = Ppu::new(PpuRegion::Ntsc);
7049        // Drive past the post-reset masking window.
7050        ppu.post_reset_mask_remaining = 0;
7051        (ppu, TestBus::new())
7052    }
7053
7054    /// v3.1.0 (`T-PAL-EMPHASIS`, ACC-01): on the PAL (2C07) and Dendy PPUs
7055    /// PPUMASK bits 5 and 6 swap meaning. `NESdev` "Colour emphasis": "Bit 5
7056    /// emphasizes red on the NTSC PPU, and green on the PAL & Dendy PPUs. Bit 6
7057    /// emphasizes green on the NTSC PPU, and red on the PAL & Dendy PPUs. Bit 7
7058    /// emphasizes blue on the NTSC, PAL, & Dendy PPUs." The emphasis index the
7059    /// renderer and the composite filters receive is the PHYSICAL tint (bit 0
7060    /// red, bit 1 green, bit 2 blue), so on PAL / Dendy it is the mask's bits
7061    /// with 5 and 6 exchanged.
7062    #[test]
7063    fn pal_and_dendy_swap_the_red_and_green_emphasis_bits() {
7064        let cases = [
7065            (PpuMask::EMPHASIZE_RED, 0b001u16, 0b010u16),
7066            (PpuMask::EMPHASIZE_GREEN, 0b010, 0b001),
7067            (PpuMask::EMPHASIZE_BLUE, 0b100, 0b100),
7068            (
7069                PpuMask::EMPHASIZE_RED | PpuMask::EMPHASIZE_BLUE,
7070                0b101,
7071                0b110,
7072            ),
7073        ];
7074        for region in [PpuRegion::Ntsc, PpuRegion::Pal, PpuRegion::Dendy] {
7075            for (mask, ntsc, swapped) in cases {
7076                let mut p = Ppu::new(region);
7077                p.post_reset_mask_remaining = 0;
7078                p.mask = mask; // rendering off: the pixel is the backdrop
7079                p.palette_ram[palette_index(0x3F00)] = 0x21;
7080                p.scanline = 10;
7081                p.dot = 20;
7082                p.emit_pixel();
7083                let got = p.index_framebuffer[10 * 256 + 19];
7084                let want_emph = if region == PpuRegion::Ntsc {
7085                    ntsc
7086                } else {
7087                    swapped
7088                };
7089                assert_eq!(
7090                    got,
7091                    (want_emph << 6) | 0x21,
7092                    "{region:?}, mask {:#04x}: emphasis index",
7093                    mask.bits()
7094                );
7095                let off = (10usize * 256 + 19) * 4;
7096                assert_eq!(
7097                    &p.framebuffer[off..off + 4],
7098                    &p.rgba_lut[usize::from(got)],
7099                    "{region:?}: the RGBA pixel follows the same index"
7100                );
7101            }
7102        }
7103    }
7104
7105    // F1.1 (Fathom accuracy remediation) — palette backdrop-override.
7106    // When rendering is disabled and the VRAM address `v` points into palette
7107    // space ($3F00-$3FFF), the palette's shared address input is driven by `v`,
7108    // so the PPU outputs the color at `v & 0x1F` INSTEAD of the universal
7109    // backdrop ($3F00). This is a display artifact only — palette RAM is never
7110    // mutated, and rendering-enabled output is unchanged. See `NESdev` "PPU
7111    // palettes"; mirrors Mesen2 `NesPpu.cpp` / ares output-stage behavior.
7112    #[test]
7113    fn palette_backdrop_override_when_rendering_disabled() {
7114        let (mut p, _b) = fresh_ppu();
7115        p.mask = PpuMask::empty(); // rendering disabled
7116        p.palette_ram[palette_index(0x3F00)] = 0x0F; // backdrop
7117        p.palette_ram[palette_index(0x3F05)] = 0x16; // override target (red)
7118        p.scanline = 10;
7119        p.dot = 20; // pixel_x = 19 (visible)
7120        let off = (10usize * 256 + 19) * 4;
7121        let red = crate::palette::nes_color_to_rgba(0x16);
7122        let backdrop = crate::palette::nes_color_to_rgba(0x0F);
7123
7124        // v in palette range, rendering off -> output palette[v & 0x1F].
7125        p.v = 0x3F05;
7126        p.emit_pixel();
7127        assert_eq!(&p.framebuffer[off..off + 4], &red, "override -> palette[5]");
7128
7129        // v NOT in palette range, rendering off -> universal backdrop.
7130        p.v = 0x2000;
7131        p.emit_pixel();
7132        assert_eq!(&p.framebuffer[off..off + 4], &backdrop, "non-palette v");
7133
7134        // Rendering ENABLED with transparent BG -> backdrop, never overridden
7135        // (the fetch pipeline owns `v` while rendering).
7136        p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_BG_LEFT;
7137        p.v = 0x3F05;
7138        p.emit_pixel();
7139        assert_eq!(
7140            &p.framebuffer[off..off + 4],
7141            &backdrop,
7142            "enabled -> no override"
7143        );
7144
7145        // $3F10 mirrors to $3F00 (universal backdrop), not a distinct entry.
7146        p.mask = PpuMask::empty();
7147        p.v = 0x3F10;
7148        p.emit_pixel();
7149        assert_eq!(
7150            &p.framebuffer[off..off + 4],
7151            &backdrop,
7152            "$3F10 mirrors backdrop"
7153        );
7154
7155        // The override is display-only: palette RAM is unmodified.
7156        assert_eq!(p.palette_ram[palette_index(0x3F05)], 0x16);
7157    }
7158
7159    // F1.2 (Fathom) — OAM / $2004 quirks. Both behaviors below are already
7160    // implemented and covered by the AccuracyCoin `$2004`/`Sprite0Hit` ROMs;
7161    // this is a FAST regression guard so an edit trips a unit test instead of
7162    // only the ~57s ROM battery. (The `OAMADDR & 0xF8` render-start copy is NOT
7163    // modeled on the DEFAULT revision — Mesen2, ares, and TriCNES all omit it as
7164    // a revision-dependent, oracle-less corner. As of v2.1.7 P5 the related
7165    // OAMADDR `$2003` write-during-render corruption is available as an opt-in
7166    // `PpuRevision::Rp2c02G` model; see `docs/accuracy-ledger.md`.)
7167    #[test]
7168    fn oam_2004_attribute_mask_and_oamaddr_257_320_forcing() {
7169        // (1) $2004 read of a sprite ATTRIBUTE byte (OAM offset & 3 == 2) masks
7170        // bits 4-2 with $E3 (they don't exist in OAM); other bytes are unmasked.
7171        // Read outside the rendering windows so the plain OAM path is taken.
7172        let (mut p, mut b) = fresh_ppu();
7173        p.mask = PpuMask::empty(); // rendering disabled
7174        p.scanline = 250; // vblank -> not a render scanline
7175        p.oam[2] = 0xFF; // attribute byte
7176        p.oam[1] = 0xFF; // tile byte (no mask)
7177        p.oam_addr = 2;
7178        assert_eq!(p.cpu_read_register(4, &mut b), 0xE3, "attr byte $E3-masked");
7179        p.oam_addr = 1;
7180        assert_eq!(
7181            p.cpu_read_register(4, &mut b),
7182            0xFF,
7183            "non-attr byte unmasked"
7184        );
7185
7186        // (2) OAMADDR is forced to 0 across dots 257-320 of a rendered scanline
7187        // (the sprite-tile-load interval), washing away a perturbed value.
7188        let (mut p, mut b) = fresh_ppu();
7189        p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
7190        p.scanline = 10; // visible render line
7191        p.dot = 256;
7192        p.oam_addr = 0x40; // perturbed; nothing but the 257-320 wash zeroes it
7193        for _ in 0..70 {
7194            p.tick(&mut b); // dot 256 -> 326, through the whole window
7195        }
7196        assert_eq!(p.oam_addr, 0, "OAMADDR washed to 0 across dots 257-320");
7197    }
7198
7199    // v2.1.4 F2.3 — optional OAM decay (opt-in, default-OFF). Models Mesen2's
7200    // `ReadSpriteRam`: a row un-refreshed for > OAM_DECAY_CPU_CYCLES CPU cycles
7201    // decays to `((sprAddr & 3) == 2) ? (sprAddr & 0xE3) : sprAddr` on the next
7202    // read. The read path used here is the plain (non-rendering) `$2004` read —
7203    // `mask` empty + a vblank scanline keeps out of the rendering / dot-1-64 /
7204    // dot-257-320 forcing windows.
7205    #[test]
7206    fn oam_decay_disabled_by_default_leaves_oam_untouched() {
7207        let (mut p, mut b) = fresh_ppu();
7208        p.mask = PpuMask::empty();
7209        p.scanline = 250; // vblank — plain OAM read path
7210        // Seed a distinctive value in row 0 (bytes 0..8).
7211        for i in 0..8u8 {
7212            p.oam[i as usize] = 0xAA;
7213        }
7214        // Advance the clock WELL past the decay window with no OAM access.
7215        p.dot_counter = (OAM_DECAY_CPU_CYCLES + 10_000) * 3;
7216        // Default is disabled — a read must return the seeded byte, and OAM must
7217        // be byte-for-byte unchanged (no decay pattern written).
7218        p.oam_addr = 0;
7219        assert!(!p.oam_decay_enabled(), "decay off by default");
7220        assert_eq!(
7221            p.cpu_read_register(4, &mut b),
7222            0xAA,
7223            "no decay when disabled"
7224        );
7225        for i in 0..8usize {
7226            assert_eq!(p.oam[i], 0xAA, "OAM row untouched when decay disabled");
7227        }
7228    }
7229
7230    #[test]
7231    fn oam_decay_enabled_decays_stale_row_to_mesen_pattern() {
7232        let (mut p, mut b) = fresh_ppu();
7233        p.set_oam_decay(true);
7234        p.mask = PpuMask::empty();
7235        p.scanline = 250;
7236        // Seed row 3 (OAM $18..$20) with a value distinct from the decay pattern.
7237        for a in 0x18u8..0x20 {
7238            p.oam[a as usize] = 0x5A;
7239        }
7240        // `set_oam_decay(true)` re-based the timestamps to the then-current cycle
7241        // (0). Advance PAST the window so the row is stale on the next read.
7242        p.dot_counter = (OAM_DECAY_CPU_CYCLES + 1) * 3;
7243        // Read byte $1A (an attribute byte: $1A & 3 == 2) — the whole row decays
7244        // first, then the (now-decayed) byte is returned. Expected decay byte for
7245        // $1A = $1A & 0xE3 = $02; but `$2004` additionally $E3-masks an attr byte
7246        // on the way out ($02 & $E3 == $02), so the observed value is $02.
7247        p.oam_addr = 0x1A;
7248        let v = p.cpu_read_register(4, &mut b);
7249        assert_eq!(v, 0x1A & 0xE3, "attribute byte decays to sprAddr & 0xE3");
7250        // The full row now holds the canonical pattern.
7251        for a in 0x18u8..0x20 {
7252            let expect = if a & 0x03 == 0x02 { a & 0xE3 } else { a };
7253            assert_eq!(p.oam[a as usize], expect, "row byte ${a:02X} decayed");
7254        }
7255    }
7256
7257    #[test]
7258    fn oam_decay_access_within_window_refreshes_row() {
7259        let (mut p, mut b) = fresh_ppu();
7260        p.set_oam_decay(true);
7261        p.mask = PpuMask::empty();
7262        p.scanline = 250;
7263        for a in 0x18u8..0x20 {
7264            p.oam[a as usize] = 0x5A;
7265        }
7266        // Touch the row just before the window closes (elapsed == threshold →
7267        // still a refresh, not a decay), which re-stamps the timestamp.
7268        p.dot_counter = OAM_DECAY_CPU_CYCLES * 3;
7269        p.oam_addr = 0x18;
7270        assert_eq!(
7271            p.cpu_read_register(4, &mut b),
7272            0x5A,
7273            "in-window read: no decay"
7274        );
7275        // Advance another (threshold) cycles from the refresh point — still within
7276        // the window relative to the refreshed timestamp, so no decay.
7277        p.dot_counter += OAM_DECAY_CPU_CYCLES * 3;
7278        p.oam_addr = 0x19;
7279        assert_eq!(
7280            p.cpu_read_register(4, &mut b),
7281            0x5A,
7282            "refresh kept row alive"
7283        );
7284        for a in 0x18u8..0x20 {
7285            assert_eq!(p.oam[a as usize], 0x5A, "row still holds seeded data");
7286        }
7287    }
7288
7289    #[test]
7290    fn oam_decay_is_pal_disabled() {
7291        // PAL's frequent refresh cadence masks decay, so the model never acts
7292        // there even when enabled (matches Mesen2). Same stale-row setup as the
7293        // NTSC decay test, but on PAL the read must NOT decay.
7294        let mut p = Ppu::new(PpuRegion::Pal);
7295        p.post_reset_mask_remaining = 0;
7296        let mut b = TestBus::new();
7297        p.set_oam_decay(true);
7298        p.mask = PpuMask::empty();
7299        p.scanline = 250;
7300        for a in 0x18u8..0x20 {
7301            p.oam[a as usize] = 0x5A;
7302        }
7303        p.dot_counter = (OAM_DECAY_CPU_CYCLES + 1) * 3;
7304        p.oam_addr = 0x18;
7305        assert_eq!(p.cpu_read_register(4, &mut b), 0x5A, "PAL: decay disabled");
7306        for a in 0x18u8..0x20 {
7307            assert_eq!(p.oam[a as usize], 0x5A, "PAL row untouched");
7308        }
7309    }
7310
7311    #[test]
7312    fn oam_decay_write_refreshes_row() {
7313        let (mut p, mut b) = fresh_ppu();
7314        p.set_oam_decay(true);
7315        p.mask = PpuMask::empty();
7316        p.scanline = 250;
7317        for a in 0x18u8..0x20 {
7318            p.oam[a as usize] = 0x5A;
7319        }
7320        // A `$2004` write of row 3 well past the window still refreshes it, so a
7321        // subsequent in-window read of that row does not decay.
7322        p.dot_counter = (OAM_DECAY_CPU_CYCLES + 5_000) * 3;
7323        p.oam_addr = 0x18;
7324        p.cpu_write_register(4, 0x33, &mut b); // writes $18, refreshes row 3
7325        // Read $19 (same row) a short time later — within the window of the write.
7326        p.dot_counter += 10 * 3;
7327        p.oam_addr = 0x19;
7328        assert_eq!(p.cpu_read_register(4, &mut b), 0x5A, "write kept row alive");
7329    }
7330
7331    // v2.1.7 P5 — PPU revision + power-up palette (opt-in, default-off).
7332
7333    #[test]
7334    fn revision_defaults_to_rp2c02h_no_corruption() {
7335        let (p, _b) = fresh_ppu();
7336        assert_eq!(p.revision(), PpuRevision::Rp2c02H, "default revision");
7337        assert!(
7338            !p.revision().models_oamaddr_corruption(),
7339            "default revision models no OAMADDR corruption"
7340        );
7341    }
7342
7343    #[test]
7344    fn default_revision_2003_write_during_render_does_not_corrupt() {
7345        // On the default revision a $2003 write mid-render must NOT arm any OAM
7346        // corruption — the byte-identity guarantee.
7347        let (mut p, mut b) = fresh_ppu();
7348        p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
7349        p.scanline = 10; // visible render line
7350        // Distinct row-0 vs row-1 so a spurious copy would be observable.
7351        for i in 0..8u8 {
7352            p.oam[i as usize] = 0x11;
7353            p.oam[8 + i as usize] = 0x22;
7354        }
7355        p.cpu_write_register(3, 0x08, &mut b); // OAMADDR = row 1
7356        assert!(
7357            !p.oam_corruption_pending,
7358            "default revision: no corruption armed"
7359        );
7360    }
7361
7362    #[test]
7363    fn rp2c02g_2003_write_during_render_arms_and_corrupts_row() {
7364        // On the earlier `Rp2c02G` revision a $2003 write while rendering is
7365        // active arms the row-copy corruption; committing it copies row 0 over
7366        // the targeted row (index = value >> 3).
7367        let (mut p, mut b) = fresh_ppu();
7368        p.set_revision(PpuRevision::Rp2c02G);
7369        p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
7370        p.scanline = 10; // visible render line
7371        for i in 0..8u8 {
7372            p.oam[i as usize] = 0x11; // row 0
7373            p.oam[8 + i as usize] = 0x22; // row 1 (target)
7374        }
7375        p.cpu_write_register(3, 0x08, &mut b); // OAMADDR = 0x08 → row index 1
7376        assert!(p.oam_corruption_pending, "Rp2c02G: corruption armed");
7377        assert_eq!(p.oam_corruption_index, 1, "targets row 1");
7378        // Commit and verify row 1 now mirrors row 0.
7379        p.process_oam_corruption();
7380        for i in 0..8u8 {
7381            assert_eq!(
7382                p.oam[8 + i as usize],
7383                0x11,
7384                "row 1 byte {i} corrupted from row 0"
7385            );
7386        }
7387    }
7388
7389    #[test]
7390    fn rp2c02g_2003_write_outside_render_does_not_corrupt() {
7391        // Even on the corrupting revision, a $2003 write with rendering disabled
7392        // (or in vblank) must NOT arm corruption — the glitch is render-gated.
7393        let (mut p, mut b) = fresh_ppu();
7394        p.set_revision(PpuRevision::Rp2c02G);
7395        p.mask = PpuMask::empty(); // rendering disabled
7396        p.scanline = 250; // vblank
7397        p.cpu_write_register(3, 0x08, &mut b);
7398        assert!(
7399            !p.oam_corruption_pending,
7400            "Rp2c02G but no rendering: no corruption"
7401        );
7402    }
7403
7404    #[test]
7405    fn power_up_palette_defaults_zeroed() {
7406        let (p, _b) = fresh_ppu();
7407        assert_eq!(p.power_up_palette(), PaletteInit::Zeroed, "default palette");
7408        assert_eq!(
7409            p.palette_ram, [0u8; 32],
7410            "default power-up palette all-zero"
7411        );
7412    }
7413
7414    #[test]
7415    fn power_up_palette_blargg_applies_masked_pattern() {
7416        let (mut p, _b) = fresh_ppu();
7417        p.apply_power_up_palette(PaletteInit::Blargg);
7418        assert_eq!(p.power_up_palette(), PaletteInit::Blargg);
7419        // Byte 0 = 0x09, an attr-index that survives the 6-bit mask untouched.
7420        assert_eq!(p.palette_ram[0], 0x09, "Blargg byte 0");
7421        // Every cell must be 6-bit masked (matching a `$2007` write path).
7422        for (i, &b) in p.palette_ram.iter().enumerate() {
7423            assert_eq!(b, BLARGG_POWER_UP_PALETTE[i] & 0x3F, "cell {i} masked");
7424        }
7425        // Re-applying Zeroed restores the byte-identical default state.
7426        p.apply_power_up_palette(PaletteInit::Zeroed);
7427        assert_eq!(p.palette_ram, [0u8; 32], "re-zeroed");
7428    }
7429
7430    // F1.3 (Fathom) — PPU open-bus refresh map. The Blargg `ppu_open_bus` table
7431    // is: a read DRIVES (and refreshes) some bits and passes others through from
7432    // the decay latch. $2000-$2003/$2005/$2006 = all decay; $2004 + $2007
7433    // (non-palette) = all driven; $2002 = `---D DDDD` (bits 7-5 driven); $2007
7434    // palette = `DD-- ----` (bits 7-6 decay). The $2002 low-5 case is covered by
7435    // `ppustatus_*` above; this locks the $2007-palette and write-only cases.
7436    /// v2.9.5 (a review of #575): on the odd-frame-deferred pixel a drawing
7437    /// sprite emits its FIRST column at X=0 whatever its own X, so the HD-pack
7438    /// tile source must record column 0 there. With X=10 the original
7439    /// `(pixel_x - x) & 7` recorded column 6.
7440    #[cfg(feature = "hd-pack")]
7441    #[test]
7442    fn hd_source_records_column_zero_on_the_deferred_pixel() {
7443        let (mut p, _b) = fresh_ppu();
7444        p.mask = PpuMask::SHOW_SPRITE | PpuMask::SHOW_SPRITE_LEFT;
7445        p.scanline = 0;
7446        p.dot = 1; // pixel 0
7447        p.spr_count = 1;
7448        p.spr_x[0] = 10;
7449        p.hd_spr_x[0] = 10;
7450        p.spr_halted[0] = true; // the drawing state the skip leaves it in
7451        p.spr_shift_lo[0] = 0x80; // an opaque first column
7452        p.spr_attr[0] = 0;
7453        p.spr_rearm_deferred = true;
7454        p.emit_pixel();
7455        let rec = p.hd_tile_source()[0];
7456        assert!(rec.is_sprite, "the deferred pixel must be the sprite's");
7457        assert_eq!(rec.offset_x, 0, "the first column is drawn at X=0");
7458        assert!(!p.spr_rearm_deferred, "released after pixel 0");
7459    }
7460
7461    /// An OAM DMA byte is a `$2004` write, and "writing any value to any PPU
7462    /// port ... will fill this latch" (`nesdev_wiki/PPU_registers`, the
7463    /// `_io_db` latch). So `$2002`'s low five bits read the last DMA byte.
7464    /// v2.9.5, the sibling's `oracle-vs-documentation.md` 3.1c: the `MiSTer`
7465    /// core already did this. The oracle did not, and every `AccuracyCoin`
7466    /// entry reports the same result either way, `Open Bus` included.
7467    #[test]
7468    fn oam_dma_byte_fills_the_io_latch() {
7469        let (mut p, mut b) = fresh_ppu();
7470        p.open_bus = 0x00;
7471        p.status = PpuStatus::empty();
7472        p.oam_dma_write(0xFF);
7473        assert_eq!(p.cpu_read_register(2, &mut b) & 0x1F, 0x1F);
7474    }
7475
7476    #[test]
7477    fn open_bus_refresh_map_2007_palette_and_write_only() {
7478        // Reading a WRITE-ONLY register drives no bits -> the full decay latch.
7479        let (mut p, mut b) = fresh_ppu();
7480        p.open_bus = 0xA5;
7481        assert_eq!(
7482            p.cpu_read_register(0, &mut b),
7483            0xA5,
7484            "$2000 read = pure open bus"
7485        );
7486
7487        // $2007 PALETTE read drives bits 5-0 (palette) and passes bits 7-6 from
7488        // open bus (Blargg map: palette = `DD-- ----`).
7489        let (mut p, mut b) = fresh_ppu();
7490        p.open_bus = 0xFF; // bits 7-6 set
7491        p.mask = PpuMask::empty(); // no render-window $FF path
7492        p.v = 0x3F00;
7493        p.palette_ram[palette_index(0x3F00)] = 0x15;
7494        assert_eq!(
7495            p.cpu_read_register(7, &mut b),
7496            0x15 | 0xC0,
7497            "$2007 palette: bits 5-0 palette, 7-6 open bus"
7498        );
7499    }
7500
7501    #[test]
7502    fn ppustatus_read_clears_vbl_and_w() {
7503        let (mut p, mut b) = fresh_ppu();
7504        p.status.insert(PpuStatus::VBLANK);
7505        p.w = true;
7506        let v = p.cpu_read_register(2, &mut b);
7507        assert!(v & 0x80 != 0, "VBL should have been set on read");
7508        assert!(!p.status.contains(PpuStatus::VBLANK));
7509        assert!(!p.w);
7510    }
7511
7512    #[test]
7513    fn default_ppu_uses_composite_palette_no_2c05() {
7514        let (p, _b) = fresh_ppu();
7515        assert_eq!(p.active_palette, crate::palette::PpuPalette::Composite2C02);
7516        assert!(!p.is_2c05);
7517        // map_register is the identity on a non-2C05 PPU.
7518        for r in 0u8..8 {
7519            assert_eq!(p.map_register(r), r);
7520        }
7521    }
7522
7523    #[test]
7524    fn c2c05_swaps_2000_and_2001() {
7525        let (mut p, mut b) = fresh_ppu();
7526        p.set_palette(crate::palette::PpuPalette::Rgb2C05, true, 0x3D);
7527        // A write to $2000 (reg 0) on a 2C05 sets MASK; a write to $2001 sets
7528        // CTRL. Use a distinct, register-valid value for each.
7529        // PPUMASK bit 3 = SHOW_BG. PPUCTRL bit 7 = NMI_ENABLE.
7530        p.cpu_write_register(0, 0b0000_1000, &mut b); // -> MASK SHOW_BG
7531        assert!(p.mask.contains(PpuMask::SHOW_BG), "$2000 write set MASK");
7532        assert!(p.ctrl.is_empty(), "$2000 write did NOT touch CTRL");
7533
7534        let (mut p2, mut b2) = fresh_ppu();
7535        p2.set_palette(crate::palette::PpuPalette::Rgb2C05, true, 0x3D);
7536        p2.cpu_write_register(1, 0b1000_0000, &mut b2); // -> CTRL NMI_ENABLE
7537        assert!(
7538            p2.ctrl.contains(PpuCtrl::NMI_ENABLE),
7539            "$2001 write set CTRL on a 2C05"
7540        );
7541        assert!(p2.mask.is_empty(), "$2001 write did NOT touch MASK");
7542    }
7543
7544    #[test]
7545    fn c2c05_2002_returns_identifier_in_low_bits() {
7546        let (mut p, mut b) = fresh_ppu();
7547        p.set_palette(crate::palette::PpuPalette::Rgb2C05, true, 0x3D);
7548        // Set the VBL flag so the high bits are deterministic.
7549        p.status.insert(PpuStatus::VBLANK);
7550        let v = p.cpu_read_register(2, &mut b);
7551        // 2C05-02 id = $3D; low 5 bits => $3D & $1F = $1D.
7552        assert_eq!(v & 0x1F, 0x3D & 0x1F);
7553        assert!(v & 0x80 != 0, "VBL still reported in bit 7");
7554    }
7555
7556    #[test]
7557    fn non_2c05_2002_keeps_open_bus_low_bits() {
7558        // Without is_2c05, the low 5 bits remain open-bus (byte-identical to
7559        // the legacy path).
7560        let (mut p, mut b) = fresh_ppu();
7561        p.cpu_write_register(3, 0x1F, &mut b); // load open bus with $1F
7562        let v = p.cpu_read_register(2, &mut b);
7563        assert_eq!(v & 0x1F, 0x1F);
7564    }
7565
7566    #[test]
7567    fn ppustatus_low_5_bits_are_open_bus() {
7568        let (mut p, mut b) = fresh_ppu();
7569        // Touch the open-bus latch via a $2003 write.
7570        p.cpu_write_register(3, 0xAB, &mut b);
7571        p.status.insert(PpuStatus::VBLANK);
7572        let v = p.cpu_read_register(2, &mut b);
7573        // Bits 7-5 from status (only VBL set), bits 4-0 from open-bus (0x0B).
7574        assert_eq!(v & 0xE0, 0x80);
7575        assert_eq!(v & 0x1F, 0xAB & 0x1F);
7576    }
7577
7578    #[test]
7579    fn ppustatus_read_preserves_low_5_bits_of_open_bus_latch() {
7580        // Reading $2002 only refreshes the upper 3 bits of the open-bus
7581        // latch (the bits sourced from PPUSTATUS); the lower 5 bits must
7582        // retain their previous value.  Required by the `open_bus_read_test`
7583        // sub-routine of `cpu_dummy_writes_ppumem.nes` (Bisqwit), which
7584        // performs `lda $2002; eor $2000` and expects the result to be 0
7585        // after AND-masking with 0x1F.
7586        let (mut p, mut b) = fresh_ppu();
7587        // Seed the open-bus latch via a $2003 write; pick a value with low
7588        // bits set so the bug-fix is observable.
7589        p.cpu_write_register(3, 0xAB, &mut b);
7590        p.status.insert(PpuStatus::VBLANK);
7591        // Read $2002 — should expose status high bits + latch low 5 bits.
7592        let v = p.cpu_read_register(2, &mut b);
7593        assert_eq!(v, 0x80 | (0xAB & 0x1F));
7594        // Now read $2000 (write-only): should return the refreshed latch
7595        // = (status & 0xE0) | (old_latch & 0x1F) — i.e., the same value.
7596        let after = p.cpu_read_register(0, &mut b);
7597        assert_eq!(
7598            after, v,
7599            "$2002 read must refresh only the high 3 bits of open-bus; \
7600             the low 5 bits must survive into subsequent reads of \
7601             write-only ports"
7602        );
7603    }
7604
7605    #[test]
7606    fn ppudata_buffered_read_returns_previous_byte() {
7607        let (mut p, mut b) = fresh_ppu();
7608        // CIRAM lives in the PPU now.
7609        p.ciram[0] = 0xAB;
7610        p.ciram[1] = 0xCD;
7611        // Set v to $2000.
7612        p.cpu_write_register(6, 0x20, &mut b);
7613        p.cpu_write_register(6, 0x00, &mut b);
7614        // First read: returns buffer (0), refills from $2000.
7615        let r1 = p.cpu_read_register(7, &mut b);
7616        assert_eq!(r1, 0);
7617        // Second read: returns refill (0xAB), refills with next byte.
7618        let r2 = p.cpu_read_register(7, &mut b);
7619        assert_eq!(r2, 0xAB);
7620        let r3 = p.cpu_read_register(7, &mut b);
7621        assert_eq!(r3, 0xCD);
7622    }
7623
7624    #[test]
7625    fn ppudata_palette_read_bypasses_buffer() {
7626        let (mut p, mut b) = fresh_ppu();
7627        p.palette_ram[0] = 0x12;
7628        // Stash a different value in the underlying nametable mirror so we
7629        // see the buffer get the underlying value, not the palette byte.
7630        p.ciram[0] = 0xCC;
7631        // Set v to $3F00.
7632        p.cpu_write_register(6, 0x3F, &mut b);
7633        p.cpu_write_register(6, 0x00, &mut b);
7634        let r = p.cpu_read_register(7, &mut b);
7635        // High 2 bits open-bus. Low 6 bits: 0x12.
7636        assert_eq!(r & 0x3F, 0x12);
7637        // Buffer should now contain underlying nametable mirror at $2F00
7638        // (= $3F00 & $2FFF), via horizontal mirroring tables 2/3 -> bank 1.
7639    }
7640
7641    #[test]
7642    fn ppudata_increment_1_or_32() {
7643        let (mut p, mut b) = fresh_ppu();
7644        p.cpu_write_register(6, 0x21, &mut b);
7645        p.cpu_write_register(6, 0x00, &mut b);
7646        // Increment by 1 default.
7647        p.cpu_read_register(7, &mut b);
7648        assert_eq!(p.v & 0x7FFF, 0x2101);
7649        // Switch to increment 32.
7650        p.cpu_write_register(0, PpuCtrl::VRAM_INCREMENT_32.bits(), &mut b);
7651        p.cpu_read_register(7, &mut b);
7652        assert_eq!(p.v & 0x7FFF, 0x2121);
7653    }
7654
7655    #[test]
7656    fn ppuctrl_post_reset_mask_window_blocks_writes() {
7657        let mut p = Ppu::new(PpuRegion::Ntsc);
7658        // Don't override post_reset_mask_remaining — it's the documented
7659        // count.
7660        let mut b = TestBus::new();
7661        p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7662        assert!(
7663            !p.ctrl.contains(PpuCtrl::NMI_ENABLE),
7664            "PPUCTRL write must be ignored during post-reset window"
7665        );
7666        // Drive past the window.
7667        for _ in 0..30_000 {
7668            p.on_cpu_cycle();
7669        }
7670        p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7671        assert!(p.ctrl.contains(PpuCtrl::NMI_ENABLE));
7672    }
7673
7674    /// v2.9.8 — `end_warmup` closes the post-reset window at once: the four
7675    /// masked registers (`$2000`/`$2001`/`$2005`/`$2006`, `NESdev` "PPU power up
7676    /// state") accept the very next write, with no CPU cycles elapsed. This is
7677    /// the PPU half of the opt-in Famicom console model.
7678    #[test]
7679    fn end_warmup_lets_masked_registers_write_immediately() {
7680        let mut p = Ppu::new(PpuRegion::Ntsc);
7681        let mut b = TestBus::new();
7682        assert_eq!(p.warmup_cycles_remaining(), 29_658);
7683        p.end_warmup();
7684        assert_eq!(p.warmup_cycles_remaining(), 0);
7685        p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7686        assert!(p.ctrl.contains(PpuCtrl::NMI_ENABLE), "$2000 accepted");
7687        // Greyscale only: a visible PPUMASK change that leaves rendering off,
7688        // so the `$2006` pair below copies `t -> v` at once.
7689        p.cpu_write_register(1, 0x01, &mut b);
7690        assert_eq!(p.mask.bits(), 0x01, "$2001 accepted");
7691        p.cpu_write_register(6, 0x21, &mut b);
7692        p.cpu_write_register(6, 0x08, &mut b);
7693        assert_eq!(p.v & 0x3FFF, 0x2108, "$2006 pair accepted");
7694        p.cpu_write_register(5, 0x08, &mut b);
7695        assert!(p.w, "$2005 toggled the write latch");
7696    }
7697
7698    #[test]
7699    fn ppuctrl_nmi_enable_during_vbl_asserts_nmi_immediately() {
7700        let (mut p, mut b) = fresh_ppu();
7701        p.status.insert(PpuStatus::VBLANK);
7702        // NMI not yet enabled => line low.
7703        assert!(!p.nmi_line);
7704        p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7705        assert!(p.nmi_line);
7706    }
7707
7708    #[test]
7709    fn ppuscroll_two_writes_load_t_and_x() {
7710        let (mut p, mut b) = fresh_ppu();
7711        p.cpu_write_register(5, 0b1010_1011, &mut b); // X = 0xAB
7712        // t bits 4-0 = X[7:3] = 0b10101 = 0x15. x = X[2:0] = 0b011 = 0x03.
7713        assert_eq!(p.t & 0x001F, 0x15);
7714        assert_eq!(p.x, 0x03);
7715        assert!(p.w);
7716        p.cpu_write_register(5, 0b0101_1100, &mut b); // Y = 0x5C
7717        // t bits 14-12 = Y[2:0] = 0b100, t bits 9-5 = Y[7:3] = 0b01011.
7718        assert_eq!((p.t >> 12) & 0x07, 0x04);
7719        assert_eq!((p.t >> 5) & 0x1F, 0x0B);
7720        assert!(!p.w);
7721    }
7722
7723    #[test]
7724    fn ppuaddr_two_writes_copy_t_to_v() {
7725        let (mut p, mut b) = fresh_ppu();
7726        p.cpu_write_register(6, 0x3F, &mut b); // high
7727        // After first write t bits 13-8 = 0x3F & 0x3F; bit 14 cleared.
7728        assert_eq!((p.t >> 8) & 0x7F, 0x3F);
7729        assert!(p.w);
7730        p.cpu_write_register(6, 0x10, &mut b); // low; copy t to v
7731        assert_eq!(p.v, 0x3F10);
7732        assert!(!p.w);
7733    }
7734
7735    #[test]
7736    fn vbl_set_and_nmi_at_scanline_241_dot_1() {
7737        let (mut p, mut b) = fresh_ppu();
7738        p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7739        // Tick until scanline 241 dot 1.
7740        // Starting at pre-render dot 0 (after construction we set scanline
7741        // = prerender_line, dot = 0). Tick advances first. We need to
7742        // reach scanline 241 dot 1. Simplest: just tick enough.
7743        let mut saw_nmi = false;
7744        for _ in 0..(341 * 263) {
7745            p.tick(&mut b);
7746            if p.nmi_line {
7747                saw_nmi = true;
7748                break;
7749            }
7750        }
7751        assert!(saw_nmi, "NMI must assert during VBlank");
7752        assert!(p.status.contains(PpuStatus::VBLANK));
7753    }
7754
7755    #[test]
7756    fn frame_complete_latch_fires_once_per_frame() {
7757        let (mut p, mut b) = fresh_ppu();
7758        // Tick a full frame's worth.
7759        let mut frames_seen = 0;
7760        for _ in 0..(341 * 262 * 2) {
7761            p.tick(&mut b);
7762            if p.take_frame_complete() {
7763                frames_seen += 1;
7764            }
7765        }
7766        assert!(frames_seen >= 2);
7767    }
7768
7769    #[test]
7770    fn index_framebuffer_mirrors_rgba_output() {
7771        // T-110-A1: the parallel palette-index framebuffer must be a faithful
7772        // index-space mirror of the RGBA framebuffer — for every emitted pixel,
7773        // `rgba_lut[index] == framebuffer[pixel]`. This is the contract that
7774        // makes the index buffer a safe, determinism-neutral output: it carries
7775        // exactly the LUT index used to produce the displayed RGBA.
7776        let (mut p, mut b) = fresh_ppu();
7777        // Enable background rendering so the full visible area is emitted.
7778        p.cpu_write_register(1, 0x08, &mut b); // PPUMASK: show background
7779        // Run two full frames so every visible pixel has been written.
7780        for _ in 0..(341 * 262 * 2) {
7781            p.tick(&mut b);
7782        }
7783        let fb = p.framebuffer();
7784        let idx = p.index_framebuffer();
7785        assert_eq!(idx.len(), FRAMEBUFFER_PIXELS);
7786        for (i, &lut_idx) in idx.iter().enumerate() {
7787            assert!((lut_idx as usize) < 512, "index in range at pixel {i}");
7788            let expected = p.rgba_lut[lut_idx as usize];
7789            assert_eq!(
7790                &fb[i * 4..i * 4 + 4],
7791                &expected,
7792                "pixel {i}: index {lut_idx} must reproduce the RGBA output"
7793            );
7794        }
7795    }
7796
7797    #[test]
7798    fn ntsc_phase_in_range_and_crawls() {
7799        // The per-frame NTSC phase must stay in 0..=2 (NTSC) and visit more than
7800        // one value across frames (the dot-crawl the filter reproduces).
7801        let (mut p, mut b) = fresh_ppu();
7802        p.cpu_write_register(1, 0x08, &mut b); // rendering on (odd-frame skip active)
7803        let mut seen = [false; 3];
7804        for _ in 0..(341 * 262 * 8) {
7805            p.tick(&mut b);
7806            if p.take_frame_complete() {
7807                let ph = p.ntsc_phase();
7808                assert!(ph <= 2, "NTSC phase {ph} must be 0..=2");
7809                seen[ph as usize] = true;
7810            }
7811        }
7812        let distinct = seen.iter().filter(|&&s| s).count();
7813        assert!(
7814            distinct >= 2,
7815            "phase must crawl across frames (saw {distinct})"
7816        );
7817    }
7818
7819    #[test]
7820    fn palette_mirrors_3f10_alias_3f00() {
7821        let (mut p, mut b) = fresh_ppu();
7822        p.cpu_write_register(6, 0x3F, &mut b);
7823        p.cpu_write_register(6, 0x10, &mut b); // v = $3F10
7824        p.cpu_write_register(7, 0x21, &mut b); // write palette
7825        // The mirror should land at index 0 (= $3F00).
7826        assert_eq!(p.palette_ram[0], 0x21);
7827        assert_eq!(p.palette_ram[0x10], 0); // not actually written
7828    }
7829
7830    #[test]
7831    fn oamdata_write_increments_oamaddr() {
7832        let (mut p, mut b) = fresh_ppu();
7833        p.oam_addr = 0x40;
7834        p.cpu_write_register(4, 0xCC, &mut b);
7835        assert_eq!(p.oam[0x40], 0xCC);
7836        assert_eq!(p.oam_addr, 0x41);
7837    }
7838
7839    /// v2.9.7 — the A12 stream the PPU reports is the HARDWARE stream, not
7840    /// MMC3's filtered view of it.
7841    ///
7842    /// Standard layout (BG at `$0000`, sprites at `$1000`), rendering on. In
7843    /// each of the eight 8-dot sprite-fetch slots (dots 257-320) the PPU reads
7844    /// two garbage nametable bytes (`$2xxx`, A12 low) and then the sprite's two
7845    /// pattern bytes (`$1xxx`, A12 high). So A12 rises **eight times per
7846    /// rendered line**: 240 visible lines plus the pre-render line give
7847    /// `8 * 241 = 1928` per NTSC frame.
7848    ///
7849    /// MMC3 sees one of those per line because its filter ignores a rise
7850    /// unless A12 was low for about three CPU cycles (nine dots). The first
7851    /// slot's rise follows the long low of the background fetches; every later
7852    /// slot's rise follows a four-dot low. The same filter, applied here to
7853    /// the raw stream, must therefore still count exactly 241.
7854    ///
7855    /// Until v2.9.7 the PPU never reported the garbage nametable reads, so it
7856    /// emitted MMC3's view (241 rises) as if it were the stream, and this test
7857    /// pinned that. Boards that count raw edges were starved eightfold:
7858    /// Acclaim's MC-ACC (a divide-by-8 on every edge), whose six local test
7859    /// games lost their status bars, and mapper 91 submapper 0, whose page
7860    /// says it counts "64 unfiltered rises of PPU A12".
7861    #[test]
7862    fn a12_reports_the_hardware_stream_and_an_mmc3_filter_still_sees_241() {
7863        struct CountingBus {
7864            chr: [u8; 0x2000],
7865            dot: u64,
7866            last_a12: bool,
7867            low_since: u64,
7868            raw_rises: u32,
7869            filtered_rises: u32,
7870        }
7871        impl PpuBus for CountingBus {
7872            fn ppu_read(&mut self, addr: u16) -> u8 {
7873                if addr < 0x2000 {
7874                    self.chr[addr as usize]
7875                } else {
7876                    0
7877                }
7878            }
7879            fn ppu_write(&mut self, addr: u16, value: u8) {
7880                if addr < 0x2000 {
7881                    self.chr[addr as usize] = value;
7882                }
7883            }
7884            fn notify_a12(&mut self, level: bool) {
7885                if level == self.last_a12 {
7886                    return;
7887                }
7888                if level {
7889                    self.raw_rises += 1;
7890                    // MMC3-style filter, in dots: low for at least 9 dots
7891                    // (three CPU cycles on NTSC).
7892                    if self.dot.saturating_sub(self.low_since) >= 9 {
7893                        self.filtered_rises += 1;
7894                    }
7895                } else {
7896                    self.low_since = self.dot;
7897                }
7898                self.last_a12 = level;
7899            }
7900            fn nametable_address(&self, addr: u16) -> u16 {
7901                let table = ((addr.wrapping_sub(0x2000)) / 0x0400) & 0x03;
7902                let local = addr & 0x03FF;
7903                let phys = u16::from(table >= 2);
7904                phys * 0x0400 + local
7905            }
7906        }
7907        let mut p = Ppu::new(PpuRegion::Ntsc);
7908        p.post_reset_mask_remaining = 0;
7909        let mut b = CountingBus {
7910            chr: [0u8; 0x2000],
7911            dot: 0,
7912            last_a12: false,
7913            low_since: 0,
7914            raw_rises: 0,
7915            filtered_rises: 0,
7916        };
7917        p.cpu_write_register(0, PpuCtrl::SPRITE_PATTERN_HIGH.bits(), &mut b);
7918        p.cpu_write_register(1, (PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE).bits(), &mut b);
7919        while !(p.scanline() == 0 && p.dot() == 0) {
7920            p.tick(&mut b);
7921            b.dot += 1;
7922        }
7923        b.raw_rises = 0;
7924        b.filtered_rises = 0;
7925        let start_frame = p.frame();
7926        while p.frame() == start_frame {
7927            p.tick(&mut b);
7928            b.dot += 1;
7929        }
7930        assert_eq!(b.raw_rises, 8 * 241, "eight A12 rises per rendered line");
7931        assert_eq!(
7932            b.filtered_rises, 241,
7933            "an MMC3-style filter still sees one rise per rendered line"
7934        );
7935    }
7936
7937    /// T-MMC3-BG-A12: with the background at `$1000` and sprites at `$0000`,
7938    /// the `NESdev` MMC3 page says the counter "should decrement on PPU cycle
7939    /// 324 of the previous scanline" -- one dot before the pattern-low
7940    /// fetch's ALE dot (325), the same convention the sprite path follows
7941    /// (its rises land at 260, the page's figure for the other arrangement).
7942    /// Until this change the background reported A12 at its READ dot, two
7943    /// dots later (326, and 6 for the line's own first tile), which put a
7944    /// rise caught at the first dot of a CPU cycle one cycle late and failed
7945    /// blargg `4-scanline_timing` at sub-test 9.
7946    #[test]
7947    fn background_a12_rises_at_the_mmc3_pages_dot_324() {
7948        struct RiseBus {
7949            chr: [u8; 0x2000],
7950            last_a12: bool,
7951            rises: alloc::vec::Vec<(i16, u16)>,
7952        }
7953        impl PpuBus for RiseBus {
7954            fn ppu_read(&mut self, addr: u16) -> u8 {
7955                if addr < 0x2000 {
7956                    self.chr[addr as usize]
7957                } else {
7958                    0
7959                }
7960            }
7961            fn ppu_write(&mut self, _addr: u16, _value: u8) {}
7962            fn notify_a12(&mut self, level: bool) {
7963                if level && !self.last_a12 {
7964                    self.rises.push((0, 0));
7965                }
7966                self.last_a12 = level;
7967            }
7968            fn nametable_address(&self, addr: u16) -> u16 {
7969                addr & 0x07FF
7970            }
7971        }
7972        let mut p = Ppu::new(PpuRegion::Ntsc);
7973        p.post_reset_mask_remaining = 0;
7974        let mut b = RiseBus {
7975            chr: [0u8; 0x2000],
7976            last_a12: false,
7977            rises: alloc::vec::Vec::new(),
7978        };
7979        p.cpu_write_register(0, PpuCtrl::BG_PATTERN_HIGH.bits(), &mut b);
7980        p.cpu_write_register(1, (PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE).bits(), &mut b);
7981        // Run to scanline 9, then record lines 9-10.
7982        while p.scanline() != 9 {
7983            p.tick(&mut b);
7984        }
7985        b.rises.clear();
7986        // `tick` advances to the next dot and then processes it, so a rise is
7987        // stamped with the position AFTER the tick that produced it.
7988        while p.scanline() != 11 {
7989            let before = b.rises.len();
7990            p.tick(&mut b);
7991            if b.rises.len() > before {
7992                *b.rises.last_mut().unwrap() = (p.scanline(), p.dot());
7993            }
7994        }
7995        // Scanline 10's first tile: the prefetch on line 9, and the line's
7996        // own fetches (first visible-tile group starts at dot 1).
7997        let prefetch: alloc::vec::Vec<u16> = b
7998            .rises
7999            .iter()
8000            .filter(|r| r.0 == 9 && r.1 > 320)
8001            .map(|r| r.1)
8002            .collect();
8003        assert_eq!(
8004            prefetch,
8005            [324, 332],
8006            "the next line's two prefetched tiles rise at 324 and 332"
8007        );
8008        // A visible line's dot 0 drives "the same CHR address that is later
8009        // used to fetch the low background tile byte starting at dot 5"
8010        // (NESdev PPU rendering, "Cycle 0"), so with the background at
8011        // `$1000` A12 is already high there.
8012        let first = b.rises.iter().find(|r| r.0 == 10).map(|r| r.1);
8013        assert_eq!(
8014            first,
8015            Some(0),
8016            "a visible line's dot 0 drives the BG CHR address"
8017        );
8018    }
8019
8020    /// T-MMC3-BG-A12: scanline 0's dot 0 drives the background CHR address
8021    /// (A12 high with the background at `$1000`) on an EVEN frame, and not on
8022    /// an odd frame whose pre-render skip "replac[es] the idle tick at the
8023    /// beginning of the first visible scanline with the last tick of the last
8024    /// dummy nametable fetch" (`NESdev` PPU rendering). This test is the
8025    /// exception's only pin: blargg `4-scanline_timing` synchronises with
8026    /// `sync_vbl_even`, so no ROM in the corpus reaches the odd-frame case,
8027    /// and removing the exception leaves both 4-scanline ROMs passing
8028    /// (measured 2026-10-05).
8029    #[test]
8030    fn scanline_0_dot_0_drives_bg_chr_only_when_the_skip_did_not_replace_it() {
8031        struct LevelBus {
8032            chr: [u8; 0x2000],
8033            level: bool,
8034        }
8035        impl PpuBus for LevelBus {
8036            fn ppu_read(&mut self, addr: u16) -> u8 {
8037                if addr < 0x2000 {
8038                    self.chr[addr as usize]
8039                } else {
8040                    0
8041                }
8042            }
8043            fn ppu_write(&mut self, _addr: u16, _value: u8) {}
8044            fn notify_a12(&mut self, level: bool) {
8045                self.level = level;
8046            }
8047            fn nametable_address(&self, addr: u16) -> u16 {
8048                addr & 0x07FF
8049            }
8050        }
8051        let mut p = Ppu::new(PpuRegion::Ntsc);
8052        p.post_reset_mask_remaining = 0;
8053        let mut b = LevelBus {
8054            chr: [0u8; 0x2000],
8055            level: false,
8056        };
8057        p.cpu_write_register(0, PpuCtrl::BG_PATTERN_HIGH.bits(), &mut b);
8058        p.cpu_write_register(1, (PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE).bits(), &mut b);
8059        let (mut skipped, mut kept) = (0, 0);
8060        for _ in 0..4 {
8061            // Run to the end of the pre-render line, remembering the parity
8062            // of the frame being completed.
8063            while !(p.scanline() == p.region.prerender_line() && p.dot() == 338) {
8064                p.tick(&mut b);
8065            }
8066            let odd = p.frame() & 1 == 1;
8067            // Step until scanline 0's dot 0 has been processed.
8068            while !(p.scanline() == 0 && p.dot() == 0) {
8069                p.tick(&mut b);
8070            }
8071            if odd {
8072                assert!(
8073                    !b.level,
8074                    "an odd frame's skip replaces dot 0: A12 stays low"
8075                );
8076                skipped += 1;
8077            } else {
8078                assert!(b.level, "an even frame's dot 0 drives the BG CHR address");
8079                kept += 1;
8080            }
8081        }
8082        assert!(skipped > 0 && kept > 0, "both parities were exercised");
8083    }
8084
8085    #[test]
8086    fn a12_transitions_notify_bus() {
8087        let (mut p, mut b) = fresh_ppu();
8088        // Set v to $1234 (A12 high), then $0234 (A12 low) — two transitions.
8089        p.cpu_write_register(6, 0x12, &mut b);
8090        p.cpu_write_register(6, 0x34, &mut b);
8091        assert_eq!(b.a12_count, 1);
8092        p.cpu_write_register(6, 0x02, &mut b);
8093        p.cpu_write_register(6, 0x34, &mut b);
8094        assert_eq!(b.a12_count, 2);
8095    }
8096
8097    // -------------------------------------------------------------------
8098    // T-23-002: sprite-evaluation FSM with buggy n+m overflow increment.
8099    // -------------------------------------------------------------------
8100
8101    /// Regression: 8 in-range sprites must populate secondary OAM and
8102    /// leave `spr_count == 8` without setting the `SPRITE_OVERFLOW` flag,
8103    /// PROVIDED the diagonal-read scan over the remaining 56 sprites
8104    /// never lands on an in-range byte. To pin that condition we fill
8105    /// the entire off-screen OAM region with 0xF0, so every byte the
8106    /// buggy `n+m` walk could land on reads as y=240 (out of range).
8107    #[test]
8108    fn sprite_eval_8_sprites_no_overflow() {
8109        let (mut p, _b) = fresh_ppu();
8110        p.scanline = 0;
8111        // 8 in-range sprites with non-zero, non-conflicting byte values
8112        // that don't read as "in-range y" if the diagonal walk hits them.
8113        for i in 0..8 {
8114            let base = i * 4;
8115            p.oam[base] = 0; // y = 0 (in range)
8116            p.oam[base + 1] = 0xF0; // tile (also out of range if read as y)
8117            p.oam[base + 2] = 0xF0;
8118            p.oam[base + 3] = 0xF0;
8119        }
8120        // Sprites 8..63: every byte = 0xF0 so diagonal read finds nothing.
8121        for i in 8..64 {
8122            for j in 0..4 {
8123                p.oam[i * 4 + j] = 0xF0;
8124            }
8125        }
8126        run_per_dot_fsm(&mut p);
8127        assert_eq!(p.spr_count, 8, "exactly 8 in-range sprites must fill");
8128        assert!(
8129            !p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8130            "8 sprites + all-off-screen-remainder is not overflow"
8131        );
8132    }
8133
8134    /// The headline case: 9 in-range sprites must set `SPRITE_OVERFLOW`.
8135    /// On real hardware the buggy `n+m` increment reads the wrong byte
8136    /// of sprite #9, but here sprite #9 is in-range and its y-byte
8137    /// (which the diagonal walk reads first at n=9, m=0 if found==8)
8138    /// is in-range, so the flag fires.
8139    #[test]
8140    fn sprite_eval_9_sprites_sets_overflow() {
8141        let (mut p, _b) = fresh_ppu();
8142        p.scanline = 0;
8143        for i in 0..9 {
8144            let base = i * 4;
8145            p.oam[base] = 0; // y = 0 (in range)
8146            p.oam[base + 1] = 0xF0; // tile (out of range as y)
8147            p.oam[base + 2] = 0xF0;
8148            p.oam[base + 3] = 0xF0;
8149        }
8150        for i in 9..64 {
8151            for j in 0..4 {
8152                p.oam[i * 4 + j] = 0xF0;
8153            }
8154        }
8155        run_per_dot_fsm(&mut p);
8156        assert_eq!(p.spr_count, 8, "secondary OAM holds first 8 only");
8157        assert!(
8158            p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8159            "9 in-range sprites must set overflow"
8160        );
8161    }
8162
8163    /// Empty OAM: no in-range sprites, no overflow.
8164    #[test]
8165    fn sprite_eval_empty_oam_no_overflow() {
8166        let (mut p, _b) = fresh_ppu();
8167        p.scanline = 0;
8168        // Every byte off-screen, so the eval pass never finds anything
8169        // and never enters overflow-detection mode.
8170        for byte in &mut p.oam {
8171            *byte = 0xF0;
8172        }
8173        run_per_dot_fsm(&mut p);
8174        assert_eq!(p.spr_count, 0);
8175        assert!(!p.status.contains(PpuStatus::SPRITE_OVERFLOW));
8176    }
8177
8178    /// The buggy `n+m` increment: when 8 sprites have been found, the
8179    /// overflow-detection FSM reads `OAM[n*4+m].y` and increments BOTH
8180    /// `n` and `m` together on each iteration. If sprite #9 is OUT of
8181    /// range but sprite #10's *non-y byte* (which the bug reads as a
8182    /// y-coordinate) happens to be in-range, the overflow flag will
8183    /// fire — that's the documented hardware quirk, not a bug in our
8184    /// FSM.
8185    ///
8186    /// Construct a case where:
8187    /// - Sprites 0..7 are in-range (fill secondary OAM, found = 8).
8188    /// - Sprite 8's y is far off-screen (y = 0xF0, normal y-read would
8189    ///   say not-in-range).
8190    /// - Sprite 9's TILE byte (byte index 1, which the buggy m=1 read
8191    ///   when n=9 lands on) is set to a value that, interpreted as y,
8192    ///   would put the sprite on the next scanline.
8193    ///
8194    /// With the buggy FSM the overflow flag fires because the diagonal
8195    /// read finds sprite 9's tile byte (= 0) as a "y" that maps to a
8196    /// row in-range for an 8-tall sprite. A correct (non-buggy) FSM
8197    /// reading sprite #8's y first would NOT fire because sprite 8 is
8198    /// out of range.
8199    ///
8200    /// This test pins the buggy behavior; flipping it to non-buggy
8201    /// would change the assertion direction.
8202    #[test]
8203    fn sprite_eval_buggy_n_plus_m_finds_diagonal_overflow() {
8204        let (mut p, _b) = fresh_ppu();
8205        p.scanline = 0;
8206        // Start with the entire OAM off-screen.
8207        for byte in &mut p.oam {
8208            *byte = 0xF0;
8209        }
8210        // Sprites 0..7 in-range with all non-y bytes off-screen.
8211        for i in 0..8 {
8212            let base = i * 4;
8213            p.oam[base] = 0; // y = 0 (in range)
8214            // bytes 1,2,3 keep the 0xF0 fill so a stray read
8215            // doesn't mis-fire the diagonal test.
8216        }
8217        // Sprite 8 y is 0xF0 (from the bulk fill) — out of range.
8218        // Sprite 9 tile byte (OAM[9*4+1]) is the second diagonal read
8219        // target (after sprite 8's y). Setting it to 0 (= in-range y)
8220        // forces the buggy FSM to fire overflow on the SECOND iteration
8221        // of the inner loop.
8222        p.oam[9 * 4 + 1] = 0;
8223        run_per_dot_fsm(&mut p);
8224        assert_eq!(p.spr_count, 8);
8225        assert!(
8226            p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8227            "buggy n+m increment must find the diagonal-read overflow at sprite 9 byte 1"
8228        );
8229    }
8230
8231    // -------------------------------------------------------------------
8232    // Sprite-eval FSM regression corpus. Originally introduced as the
8233    // parallel-implementation firewall gating the B8 swap from single-
8234    // shot to per-dot FSM. The single-shot collapse was removed in B8c;
8235    // these tests are now the regression net pinning the FSM's observable
8236    // output against a straight-line reference implementation
8237    // (`reference_eval`).
8238    //
8239    // The corpus targets:
8240    //   - Empty OAM (no in-range)
8241    //   - Exactly 8 in-range (no overflow)
8242    //   - 9+ in-range (clean overflow)
8243    //   - Diagonal-read scenarios (sprite N out-of-range, sprite (N+k)'s
8244    //     non-y byte in-range)
8245    //   - 8x8 + 8x16 sprite heights
8246    //   - Boundary scanlines (0, 1, 239, prerender)
8247    //
8248    // Random fuzz + structured edge cases combined give 1013 cases.
8249    // -------------------------------------------------------------------
8250
8251    /// Tiny xorshift PRNG so the test is hermetic (no `rand` dep).
8252    struct XorShift(u64);
8253    impl XorShift {
8254        const fn new(seed: u64) -> Self {
8255            Self(if seed == 0 {
8256                0xDEAD_BEEF_CAFE_BABE
8257            } else {
8258                seed
8259            })
8260        }
8261        const fn next_u64(&mut self) -> u64 {
8262            let mut x = self.0;
8263            x ^= x << 13;
8264            x ^= x >> 7;
8265            x ^= x << 17;
8266            self.0 = x;
8267            x
8268        }
8269        fn next_u8(&mut self) -> u8 {
8270            (self.next_u64() & 0xFF) as u8
8271        }
8272    }
8273
8274    /// Snapshot of the observable post-dot-256 state for equivalence
8275    /// comparison.
8276    #[derive(Debug, Clone, PartialEq, Eq)]
8277    struct EvalObservable {
8278        secondary_oam: [u8; 32],
8279        spr_count: u8,
8280        spr_zero_in_line: bool,
8281        overflow: bool,
8282    }
8283
8284    fn observe(p: &Ppu) -> EvalObservable {
8285        EvalObservable {
8286            secondary_oam: p.secondary_oam,
8287            spr_count: p.spr_count,
8288            spr_zero_in_line: p.spr_zero_in_line,
8289            overflow: p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8290        }
8291    }
8292
8293    /// Build a fresh PPU and seed `oam`, `scanline`, and `ctrl` from the
8294    /// given parameters.
8295    fn build_case(oam: &[u8; 256], scanline: i16, ctrl: PpuCtrl) -> Ppu {
8296        let mut p = Ppu::new(PpuRegion::Ntsc);
8297        p.post_reset_mask_remaining = 0;
8298        p.oam.copy_from_slice(oam);
8299        p.scanline = scanline;
8300        p.ctrl = ctrl;
8301        // Reset the overflow flag so we can observe per-case sets.
8302        p.status.remove(PpuStatus::SPRITE_OVERFLOW);
8303        // Pre-fill secondary OAM with a poison value so the per-dot FSM's
8304        // clear phase is observable (single-shot also starts by writing
8305        // $FF into all 32 bytes, so the final state must match).
8306        p.secondary_oam = [0xAA; 32];
8307        p.spr_count = 0;
8308        p.spr_zero_in_line = false;
8309        p
8310    }
8311
8312    /// Drive the per-dot FSM through dots 0..=256 on `p`.
8313    fn run_per_dot_fsm(p: &mut Ppu) {
8314        for dot in 0..=256u16 {
8315            p.dot = dot;
8316            p.tick_sprite_eval_per_dot();
8317        }
8318    }
8319
8320    /// Run one case through the FSM and assert observable matches the
8321    /// expected pinned state. The expected state is built by computing
8322    /// the result in a non-buggy reference implementation (the
8323    /// `reference_eval` below).
8324    fn assert_case_matches(label: &str, oam: &[u8; 256], scanline: i16, ctrl: PpuCtrl) {
8325        let expected = reference_eval(oam, scanline, ctrl);
8326
8327        let mut pf = build_case(oam, scanline, ctrl);
8328        run_per_dot_fsm(&mut pf);
8329        let actual = observe(&pf);
8330
8331        assert_eq!(
8332            expected,
8333            actual,
8334            "FSM mismatch for case `{label}` \
8335             (scanline={scanline}, 8x16={}, sprite_zero_y={:#04x})",
8336            ctrl.contains(PpuCtrl::SPRITE_SIZE_16),
8337            oam[0],
8338        );
8339    }
8340
8341    /// Reference implementation: a straight-line sprite-eval emulation
8342    /// matching the 2C02's behavior, used as the golden expected output
8343    /// for the FSM regression corpus. Originally the FSM was validated
8344    /// against the old single-shot collapse via the 1013-case equivalence
8345    /// harness (B8a); after B8c removed the single-shot, this stand-alone
8346    /// reference plays the same role.
8347    fn reference_eval(oam: &[u8; 256], scanline: i16, ctrl: PpuCtrl) -> EvalObservable {
8348        // Y-test convention: see `tick_sprite_eval_per_dot` docstring.
8349        // Pre-render uses -1 (always-fail), visible uses the current
8350        // scanline; sprite Y=N renders on scanlines N+1..=N+h.
8351        let next_line: i16 = if scanline == PpuRegion::Ntsc.prerender_line() {
8352            -1
8353        } else {
8354            scanline
8355        };
8356        let sprite_height: i16 = if ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
8357            16
8358        } else {
8359            8
8360        };
8361
8362        let mut secondary_oam = [0xFFu8; 32];
8363        let mut found = 0u8;
8364        let mut spr_zero_in_line = false;
8365        let mut overflow = false;
8366
8367        let mut n_idx = 0usize;
8368        while n_idx < 64 {
8369            let base = n_idx * 4;
8370            let y = oam[base] as i16;
8371            let row = next_line - y;
8372            if row >= 0 && row < sprite_height {
8373                let sec_base = (found as usize) * 4;
8374                secondary_oam[sec_base] = oam[base];
8375                secondary_oam[sec_base + 1] = oam[base + 1];
8376                secondary_oam[sec_base + 2] = oam[base + 2];
8377                secondary_oam[sec_base + 3] = oam[base + 3];
8378                if n_idx == 0 {
8379                    spr_zero_in_line = true;
8380                }
8381                found += 1;
8382                if found == 8 {
8383                    n_idx += 1;
8384                    let mut m = 0u8;
8385                    while n_idx < 64 {
8386                        let nb = n_idx * 4 + (m as usize);
8387                        let by = oam[nb] as i16;
8388                        let brow = next_line - by;
8389                        if brow >= 0 && brow < sprite_height {
8390                            overflow = true;
8391                            break;
8392                        }
8393                        m = (m + 1) & 0x03;
8394                        n_idx += 1;
8395                    }
8396                    break;
8397                }
8398            }
8399            n_idx += 1;
8400        }
8401
8402        EvalObservable {
8403            secondary_oam,
8404            spr_count: found,
8405            spr_zero_in_line,
8406            overflow,
8407        }
8408    }
8409
8410    #[test]
8411    fn sprite_fsm_equivalence_edge_cases() {
8412        // 1: empty OAM (all 0xFF y) -> no found, no overflow.
8413        let mut oam = [0xFFu8; 256];
8414        assert_case_matches("empty_oam_y_ff", &oam, 0, PpuCtrl::empty());
8415
8416        // 2: every byte 0xF0 (out of range) -> no found, no overflow.
8417        oam = [0xF0u8; 256];
8418        assert_case_matches("empty_oam_y_f0", &oam, 0, PpuCtrl::empty());
8419
8420        // 3: 8 in-range sprites, all other bytes 0xF0 -> 8 found, no
8421        // overflow.
8422        oam = [0xF0u8; 256];
8423        for i in 0..8 {
8424            oam[i * 4] = 0;
8425        }
8426        assert_case_matches("8_in_range", &oam, 0, PpuCtrl::empty());
8427
8428        // 4: 9 in-range sprites -> overflow set.
8429        oam = [0xF0u8; 256];
8430        for i in 0..9 {
8431            oam[i * 4] = 0;
8432        }
8433        assert_case_matches("9_in_range", &oam, 0, PpuCtrl::empty());
8434
8435        // 5: 8 in-range + diagonal-read overflow (sprite 9 byte 1 = 0
8436        // forces buggy n+m to fire).
8437        oam = [0xF0u8; 256];
8438        for i in 0..8 {
8439            oam[i * 4] = 0;
8440        }
8441        oam[9 * 4 + 1] = 0;
8442        assert_case_matches("diagonal_overflow", &oam, 0, PpuCtrl::empty());
8443
8444        // 6: 8x16 sprite mode.
8445        oam = [0xF0u8; 256];
8446        for i in 0..3 {
8447            oam[i * 4] = 0;
8448        }
8449        assert_case_matches("8x16_mode", &oam, 0, PpuCtrl::SPRITE_SIZE_16);
8450
8451        // 7: pre-render line (evaluates for scanline 0).
8452        oam = [0xF0u8; 256];
8453        for i in 0..5 {
8454            oam[i * 4] = 0;
8455        }
8456        let prerender = PpuRegion::Ntsc.prerender_line();
8457        assert_case_matches("prerender_line", &oam, prerender, PpuCtrl::empty());
8458
8459        // 8: last visible scanline.
8460        oam = [0xF0u8; 256];
8461        for i in 0..2 {
8462            oam[i * 4] = 239;
8463        }
8464        assert_case_matches("scanline_239", &oam, 238, PpuCtrl::empty());
8465
8466        // 9: sprite zero NOT in range -> spr_zero_in_line must stay false.
8467        oam = [0xF0u8; 256];
8468        oam[0] = 0xF0; // sprite 0 out of range
8469        for i in 1..3 {
8470            oam[i * 4] = 0;
8471        }
8472        assert_case_matches("zero_out_of_range", &oam, 0, PpuCtrl::empty());
8473
8474        // 10: sprite zero in range but not first -> still must be true
8475        // because sprite 0 is at OAM index 0.
8476        oam = [0xF0u8; 256];
8477        oam[0] = 0; // sprite 0 in range
8478        for i in 5..10 {
8479            oam[i * 4] = 0;
8480        }
8481        assert_case_matches("zero_in_range_plus_others", &oam, 0, PpuCtrl::empty());
8482
8483        // 11: exactly 1 in-range at the last possible sprite (sprite 63).
8484        oam = [0xF0u8; 256];
8485        oam[63 * 4] = 0;
8486        assert_case_matches("only_sprite_63", &oam, 0, PpuCtrl::empty());
8487
8488        // 12: 8 in-range scattered among the 64 entries.
8489        oam = [0xF0u8; 256];
8490        for (slot, &n) in [0u8, 5, 11, 18, 27, 35, 44, 55].iter().enumerate() {
8491            let _ = slot;
8492            oam[(n as usize) * 4] = 0;
8493        }
8494        assert_case_matches("8_scattered", &oam, 0, PpuCtrl::empty());
8495
8496        // 13: all 64 sprites in range -> 8 found + overflow.
8497        oam = [0u8; 256];
8498        for i in 0..64 {
8499            oam[i * 4] = 0; // y = 0
8500            oam[i * 4 + 1] = 0xAB;
8501            oam[i * 4 + 2] = 0xCD;
8502            oam[i * 4 + 3] = 0xEF;
8503        }
8504        assert_case_matches("all_64_in_range", &oam, 0, PpuCtrl::empty());
8505    }
8506
8507    #[test]
8508    fn sprite_fsm_equivalence_randomized_corpus() {
8509        // 1000 fully-random cases + the 13 edge cases above = 1013 total
8510        // regression checks. Each invocation runs the FSM on a random
8511        // OAM/scanline/ctrl seed and asserts observable equality with
8512        // the straight-line reference implementation.
8513        const N: usize = 1000;
8514        let mut rng = XorShift::new(0x1234_5678_9ABC_DEF0);
8515
8516        for case in 0..N {
8517            let mut oam = [0u8; 256];
8518            for b in &mut oam {
8519                *b = rng.next_u8();
8520            }
8521            // Choose scanline from {0..=239, prerender=261}. Use a bias
8522            // toward 0..=239 since that's the realistic case.
8523            let r = rng.next_u64();
8524            let scanline: i16 = if r.trailing_zeros() >= 5 {
8525                PpuRegion::Ntsc.prerender_line()
8526            } else {
8527                ((r >> 8) & 0xFF) as i16 % 240
8528            };
8529            // 8x16 mode in 1/4 of cases.
8530            let ctrl = if rng.next_u64().trailing_zeros() >= 2 {
8531                PpuCtrl::SPRITE_SIZE_16
8532            } else {
8533                PpuCtrl::empty()
8534            };
8535
8536            let expected = reference_eval(&oam, scanline, ctrl);
8537
8538            let mut pf = build_case(&oam, scanline, ctrl);
8539            run_per_dot_fsm(&mut pf);
8540            let actual = observe(&pf);
8541
8542            assert_eq!(
8543                expected,
8544                actual,
8545                "FSM regressed against reference at case #{case} \
8546                 (scanline={scanline}, 8x16={}, oam[0]={:#04x})",
8547                ctrl.contains(PpuCtrl::SPRITE_SIZE_16),
8548                oam[0],
8549            );
8550        }
8551    }
8552
8553    /// Cascade A reproducer V3: mimics `AccuracyCoin`'s
8554    /// `VerifySpriteZeroHits` step 2 (the version that EXPECTS a hit).
8555    /// Sprite 0 at Y=5 X=8 tile $C0. BG tile $C0 at nametable $2C21
8556    /// (NT 3 col 1 row 1). v = $2C00.
8557    ///
8558    /// Tile $C0 has a SINGLE opaque pixel at (col=0, row=0). With v=$2C00,
8559    /// BG tile at NT 3 position $21 displays at screen pixels (8, 8).
8560    /// Sprite at (Y=5, X=8) tile $C0 draws at scanline 6 (per nesdev:
8561    /// sprite occupies scanlines Y+1..Y+8). Sprite tile $C0's only opaque
8562    /// pixel is (col 0, row 0) → screen (8, 6).
8563    ///
8564    /// Sprite (8, 6) vs BG (8, 8) — NO geometric overlap. The test asserts
8565    /// a hit IS expected here, which is impossible without sprite Y
8566    /// semantics being different from what nesdev documents. This unit
8567    /// test makes the discrepancy concrete so it can be investigated
8568    /// against Mesen2 or other reference emulators.
8569    #[test]
8570    fn cascade_a_verify_sprite_zero_hits_step2() {
8571        let (mut p, mut b) = fresh_ppu();
8572        // Pin the PPU to (prerender, dot=0) so this diagnostic harness runs
8573        // through exactly one frame starting from the prerender boundary.
8574        // Required because Ppu::new() now starts at (prerender, dot=340)
8575        // per Session-13 Option B (close the +344-dot offset vs Mesen2);
8576        // without this reset the test's "advance one frame" loop would begin
8577        // mid-prerender and the sprite-zero-hit window would shift relative
8578        // to the BG-pipeline cycle-9 reload point this test was designed to
8579        // characterise (see docs/audit/cascade-a-investigation-2026-05-19.md
8580        // and docs/audit/session-13-cpu-boot-fix-2026-05-21.md).
8581        p.dot = 0;
8582        let tile_c0_base = 0xC0 * 16;
8583        // Tile $C0: only the (col 0, row 0) pixel is opaque (lo=$80 hi=$80).
8584        b.chr[tile_c0_base] = 0x80;
8585        b.chr[tile_c0_base + 8] = 0x80;
8586        // Tile $24: fully transparent (all-zero bytes already).
8587        // Fill NT 3 (bank 1 of CIRAM, horizontal mirroring) with $24, then
8588        // write $C0 at position $21.
8589        for i in 0..0x400 {
8590            p.ciram[0x400 + i] = 0x24;
8591        }
8592        p.ciram[0x400 + 0x021] = 0xC0;
8593        // OAM page mimics OAM DMA from a $FF-cleared page + sprite 0 init.
8594        for i in 0..256 {
8595            p.oam[i] = 0xFF;
8596        }
8597        p.oam[0] = 0x05; // Y = 5 (step 2)
8598        p.oam[1] = 0xC0; // CHR
8599        p.oam[2] = 0x03; // ATT
8600        p.oam[3] = 0x08; // X = 8
8601        // v = $2C00 (NT 3 top-left).
8602        p.v = 0x2C00;
8603        p.t = 0x2C00;
8604        // PPUCTRL = 0 (both pattern tables at $0000).
8605        p.ctrl = PpuCtrl::empty();
8606        // Enable rendering.
8607        let mask = PpuMask::SHOW_BG
8608            | PpuMask::SHOW_SPRITE
8609            | PpuMask::SHOW_BG_LEFT
8610            | PpuMask::SHOW_SPRITE_LEFT;
8611        p.mask = mask;
8612        p.mask_skip_pipe1 = mask;
8613        p.mask_for_skip_check = mask;
8614        p.status = PpuStatus::empty();
8615        // Advance ~1 full frame to allow sprite-zero hit to fire if it should.
8616        for _ in 0..(262 * 341) {
8617            p.tick(&mut b);
8618        }
8619        let hit = p.status.contains(PpuStatus::SPRITE_ZERO_HIT);
8620        // POST-FIX EXPECTATION: with the cycle-9 reload + post-emit shift
8621        // BG-pipeline correction landed (see
8622        // `docs/audit/cascade-a-investigation-2026-05-19.md`), tile $C0's
8623        // single opaque BG pixel lands at screen column 8 (PPU dot 9 of
8624        // scanline 6), exactly overlapping the sprite-zero opaque pixel
8625        // at (8, 6) → SPRITE-ZERO HIT must fire.
8626        //
8627        // The test ROM's geometry: sprite Y=5 X=8 tile $C0 has its only
8628        // opaque pixel at sprite-local (col 0, row 0) → screen (8, 6).
8629        // BG tile $C0 at NT 3 position $21 with v=$2C00 (fine Y=2,
8630        // coarse Y=0) renders at scanline 6, screen column 8, with its
8631        // only opaque pixel matching → overlap → hit.
8632        assert!(
8633            hit,
8634            "BG-pipeline fix regression: sprite-zero hit must fire for \
8635             VerifySpriteZeroHits step 2 (BG opaque at (8,6) overlaps \
8636             sprite-zero opaque at (8,6)) — see \
8637             docs/audit/cascade-a-investigation-2026-05-19.md."
8638        );
8639    }
8640
8641    /// Cascade A reproducer V2: start in VBL, enable rendering via the
8642    /// CPU-visible `$2001` write (with the 2-PPU-clock pipeline delay), do
8643    /// OAM DMA via the CPU-visible `$2003 + $2004` writes, then advance
8644    /// past pre-render → scanline 0 → scanline 1. More faithful to the
8645    /// real ROM execution path than the V1 reproducer.
8646    #[test]
8647    fn cascade_a_sprite_zero_hit_y0_x8_via_register_writes() {
8648        let (mut p, mut b) = fresh_ppu();
8649        // Load tile $FC into pattern table 0 fully-opaque.
8650        let tile_fc_base = 0xFC * 16;
8651        for row in 0..8 {
8652            b.chr[tile_fc_base + row] = 0xFF;
8653            b.chr[tile_fc_base + 8 + row] = 0x00;
8654        }
8655        // Write nametable $2001 = $FC via $2006 + $2007 (CPU-visible path).
8656        p.cpu_write_register(6, 0x20, &mut b); // hi
8657        p.cpu_write_register(6, 0x01, &mut b); // lo (v = $2001)
8658        p.cpu_write_register(7, 0xFC, &mut b);
8659        // Reset scroll: v = $2000 via $2006 + $2006.
8660        p.cpu_write_register(6, 0x20, &mut b);
8661        p.cpu_write_register(6, 0x00, &mut b);
8662        // Mimic the ROM's OAM page: ClearPage2 fills with $FF, then
8663        // InitializeSpriteZero writes sprite 0. So OAM[0..4] is the sprite,
8664        // OAM[4..256] is $FF (Y=$FF -> off-screen).
8665        for i in 0..256 {
8666            p.oam[i] = 0xFF;
8667        }
8668        // OAM DMA: write sprite 0 via $2003 (OAMADDR) + $2004 (OAMDATA).
8669        p.cpu_write_register(3, 0x00, &mut b); // OAMADDR = 0
8670        p.cpu_write_register(4, 0x00, &mut b); // sprite 0 Y = 0
8671        p.cpu_write_register(4, 0xFC, &mut b); // sprite 0 CHR = $FC
8672        p.cpu_write_register(4, 0x00, &mut b); // sprite 0 ATT = 0
8673        p.cpu_write_register(4, 0x08, &mut b); // sprite 0 X = 8
8674        // Advance to scanline 241 dot 1 (VBL start) — matches the ROM
8675        // post-WaitForVBlank position.
8676        while !(p.scanline == 241 && p.dot == 1) {
8677            p.tick(&mut b);
8678        }
8679        // Enable rendering via $2001 write (BG + SPR + show-left).
8680        let mask = (PpuMask::SHOW_BG
8681            | PpuMask::SHOW_SPRITE
8682            | PpuMask::SHOW_BG_LEFT
8683            | PpuMask::SHOW_SPRITE_LEFT)
8684            .bits();
8685        p.cpu_write_register(1, mask, &mut b);
8686        // PPUSTATUS may have VBL set; clear sprite-zero-hit start clean.
8687        p.status.remove(PpuStatus::SPRITE_ZERO_HIT);
8688        // Now advance through ~30 scanlines (rest of VBL + pre-render +
8689        // visible 0-9), matching what Clockslide_3000 covers in the ROM.
8690        for _ in 0..(30 * 341) {
8691            p.tick(&mut b);
8692        }
8693        assert!(
8694            p.status.contains(PpuStatus::SPRITE_ZERO_HIT),
8695            "Expected sprite-zero hit set after 30 scanlines past VBL. \
8696             Actual status=0x{:02X}, scanline={}, dot={}, \
8697             spr_count={}, spr_zero_in_line={}, \
8698             spr_x[0]={}, spr_shift_lo[0]=0x{:02X}, spr_shift_hi[0]=0x{:02X}, \
8699             mask=0x{:02X}, ctrl=0x{:02X}",
8700            p.status.bits(),
8701            p.scanline,
8702            p.dot,
8703            p.spr_count,
8704            p.spr_zero_in_line,
8705            p.spr_x[0],
8706            p.spr_shift_lo[0],
8707            p.spr_shift_hi[0],
8708            p.mask.bits(),
8709            p.ctrl.bits(),
8710        );
8711    }
8712
8713    /// Cascade A reproducer: the exact `AccuracyCoin TEST_Sprite0Hit_Behavior`
8714    /// sub-test 1 scenario, constructed directly without going through the
8715    /// CPU/test-ROM.
8716    ///
8717    /// Setup (matches `AccuracyCoin.asm:PREP_SpriteZeroHit` + the test's
8718    /// pre-state):
8719    ///
8720    /// - Sprite 0: `Y=$00, CHR=$FC, ATT=$00, X=$08`.
8721    /// - BG nametable: `vram[$2001] = $FC` (tile $FC at col=1, row=0).
8722    /// - CHR pattern table 0, tile $FC, all 8 rows: `lo=$FF / hi=$00`
8723    ///   (fully opaque pixels of palette colour 1).
8724    /// - `PPUMASK = $1E` (BG + SPR + `BG_LEFT` + grayscale; the actual
8725    ///   `PPUMASK_COPY` value the diagnostic probe in
8726    ///   `crates/rustynes-test-harness/src/accuracy_coin.rs` captures at frame
8727    ///   3393 — see `docs/audit/accuracycoin-readme-analysis-2026-05-17.md`
8728    ///   §"Addendum (2026-05-19, session 5)").
8729    /// - `PPUCTRL = $00` (BG and sprite pattern tables both at `$0000`).
8730    /// - `v = $2000` (top-left of nametable 0).
8731    ///
8732    /// **Expected**: sprite zero hit (PPUSTATUS bit 6) is set by the end
8733    /// of scanline 1 — sprite pixel (8..15, 1) overlaps BG pixel (8..15,
8734    /// 1) and both are opaque.
8735    ///
8736    /// **Current (2026-05-19, pre-fix)**: this test FAILS. The
8737    /// diagnostic probe shows PPUSTATUS bit 6 = 0 in the live battery
8738    /// (full ROM run). This unit test is the isolated reproducer.
8739    #[test]
8740    fn cascade_a_sprite_zero_hit_y0_x8_tile_fc_overlap() {
8741        let (mut p, mut b) = fresh_ppu();
8742        // 1. Load tile $FC into pattern table 0 with fully-opaque pixels.
8743        let tile_fc_base = 0xFC * 16;
8744        for row in 0..8 {
8745            b.chr[tile_fc_base + row] = 0xFF; // lo plane (palette bit 0)
8746            b.chr[tile_fc_base + 8 + row] = 0x00; // hi plane (palette bit 1)
8747        }
8748        // 2. Write tile $FC into nametable position $2001 (col=1, row=0).
8749        //    CIRAM bank 0 directly (horizontal mirroring: $2000-$23FF -> ciram[0..0x400]).
8750        p.ciram[0x001] = 0xFC;
8751        // 3. Sprite 0: Y=$00, CHR=$FC, ATT=$00, X=$08.
8752        p.oam[0] = 0x00;
8753        p.oam[1] = 0xFC;
8754        p.oam[2] = 0x00;
8755        p.oam[3] = 0x08;
8756        // 4. PPUMASK = SHOW_BG | SHOW_SPRITE | SHOW_BG_LEFT | grayscale.
8757        let mask_bits = PpuMask::SHOW_BG
8758            | PpuMask::SHOW_SPRITE
8759            | PpuMask::SHOW_BG_LEFT
8760            | PpuMask::SHOW_SPRITE_LEFT;
8761        p.mask = mask_bits;
8762        // Pipeline the mask through the two skip-check stages so the
8763        // rendering-enabled signal is stable immediately.
8764        p.mask_skip_pipe1 = mask_bits;
8765        p.mask_for_skip_check = mask_bits;
8766        // 5. PPUCTRL = 0 (BG and sprite both at pattern table 0).
8767        p.ctrl = PpuCtrl::empty();
8768        // 6. v = $2000 (top-left of nametable 0).
8769        p.v = 0x2000;
8770        // Make sure sprite-zero-hit and VBL start clean.
8771        p.status = PpuStatus::empty();
8772        // Pre-render starts; advance ~3 full scanlines so we cross
8773        // pre-render → scanline 0 → scanline 1 → scanline 2. By the end
8774        // of scanline 1, the sprite-zero hit should be set.
8775        // Frame is 262*341 dots. We need at least scanlines 261..=2 = 4
8776        // scanlines = 4*341 = 1364 dots. Use 1500 for safety.
8777        for _ in 0..1500 {
8778            p.tick(&mut b);
8779        }
8780        assert!(
8781            p.status.contains(PpuStatus::SPRITE_ZERO_HIT),
8782            "Expected sprite-zero hit (PPUSTATUS bit 6) to be set after \
8783             scanline 1 with sprite 0 at (Y=0, X=8) tile $FC overlapping \
8784             BG nametable[$2001]=$FC (both fully opaque). \
8785             Actual status=0x{:02X}, scanline={}, dot={}, \
8786             spr_count={}, spr_zero_in_line={}, \
8787             spr_x[0]={}, spr_shift_lo[0]=0x{:02X}, spr_shift_hi[0]=0x{:02X}",
8788            p.status.bits(),
8789            p.scanline,
8790            p.dot,
8791            p.spr_count,
8792            p.spr_zero_in_line,
8793            p.spr_x[0],
8794            p.spr_shift_lo[0],
8795            p.spr_shift_hi[0],
8796        );
8797    }
8798
8799    // =========================================================
8800    // $2002 VBL race-window sweep — Mesen2-independent oracle
8801    // (Session-18 / C1 attempt 16, PPU axis).
8802    //
8803    // The nesdev wiki [`PPU registers`] page documents the race:
8804    //
8805    //   "Reading the status register within two cycles of when VBL is
8806    //    set will return 0 in bit 7 but clear the latch anyway, causing
8807    //    the program to miss frames."
8808    //
8809    //   "Reading PPUSTATUS at the exact start of vertical blank will
8810    //    return 0 in bit 7 but clear the latch anyway, causing NMI to
8811    //    not occur that frame."
8812    //
8813    // Three documented dot-cohorts straddling scanline 241 dot 1:
8814    //
8815    //   * dot < the-VBL-set-dot (i.e. dot 0 of scanline 241, or
8816    //     earlier): VBL bit is 0 in PPUSTATUS, latch was never set,
8817    //     suppression DOES happen if read lands on dot 0 of scanline
8818    //     241 (the one-dot-before window).
8819    //   * dot == the-VBL-set-dot (= dot 1 of scanline 241): the
8820    //     "exact start of VBL" window — read returns 0, latch is
8821    //     cleared, and the in-frame VBL set is suppressed.
8822    //   * dot > the-VBL-set-dot (dot 2 or later of scanline 241):
8823    //     read returns 1 (VBL was set), latch is cleared by the
8824    //     read, no suppression of subsequent VBL/NMI within that
8825    //     frame because the set already happened.
8826    //
8827    // This unit test sweeps the PPU position across that boundary
8828    // (scanline 240 dot 339 through scanline 241 dot 5) and tabulates
8829    // the four observables per scenario: (a) the read return value's
8830    // bit 7, (b) whether suppress_vbl_this_frame got set, (c) whether
8831    // PPUSTATUS.VBLANK is set inside the PPU after the read, (d) the
8832    // value the next read of $2002 returns once we tick past dot 1.
8833    //
8834    // The test asserts the expected race-window semantics for ALL
8835    // dot positions. If `RustyNES` honours the nesdev spec, every
8836    // assertion passes. If not, the failing rows expose the exact
8837    // boundary off-by-one.
8838    //
8839    // After the test the table itself is `println!`'d for human
8840    // inspection via `--nocapture`.
8841    /// Loop budget: one full NTSC frame's worth of dots plus a
8842    /// 1024-dot safety margin, ample to sweep into scanline 242.
8843    #[cfg(test)]
8844    const VBL_SWEEP_MAX_TICKS: u32 = 262 * 341 + 1024;
8845
8846    #[test]
8847    #[allow(clippy::too_many_lines)]
8848    #[allow(clippy::items_after_statements)]
8849    fn vbl_race_window_2002_read_sweep() {
8850        use alloc::format;
8851        use alloc::string::String;
8852        use alloc::vec::Vec;
8853        // `eprintln!` lives in std; tests run in a `std` cargo unit so
8854        // this is fine.
8855        extern crate std;
8856        use std::eprintln;
8857        // The window we sweep, in (scanline, dot) pairs, listed in
8858        // tick-order. We use NTSC (vblank_start_line = 241).
8859        //
8860        // Layout choice: scan two extra dots into scanline 240 (the
8861        // last visible line) so the "VBL never gets set this frame"
8862        // pre-window is observable; then sweep dots 0..=5 of scanline
8863        // 241; then sweep two dots into scanline 242 for the post-VBL
8864        // tail. Total = 2 + 6 + 2 = 10 sample points.
8865        //
8866        // We re-create a fresh PPU for each sample-point so the
8867        // suppression-latch carries no contamination from the prior
8868        // sample. The PPU's internal state between samples is the
8869        // confounding factor we MUST isolate.
8870
8871        #[derive(Debug, Clone, Copy)]
8872        struct ExpectedRow {
8873            scanline: i16,
8874            dot: u16,
8875            // Bit 7 of the value returned by the $2002 read.
8876            // None = no specific spec assertion (don't enforce).
8877            read_bit7: Option<u8>,
8878            // Whether `suppress_vbl_this_frame` should be set after
8879            // the read. None = don't enforce.
8880            suppress_set: Option<bool>,
8881            // Whether `status.VBLANK` is set after the read (the
8882            // read always clears it, so this should be `false` for
8883            // any cohort where the read happens AT or AFTER the set
8884            // dot; and `false` for cohorts where the set never
8885            // happened either).
8886            vblank_after_read: Option<bool>,
8887        }
8888
8889        // Per the wiki, the VBL flag is set at scanline 241 dot 1.
8890        // The "race window" is documented as:
8891        //   - dot 0 of scanline 241: read returns 0, suppresses VBL set
8892        //   - dot 1 of scanline 241: read returns 0, suppresses VBL set
8893        //   - dot 2 of scanline 241: read returns 1, normal clear
8894        //
8895        // RustyNES's current impl reads back `dot <= 1` for the
8896        // suppression branch (see `cpu_read_register` case 2 above:
8897        // `if self.scanline == self.region.vblank_start_line() &&
8898        // self.dot <= 1 { self.suppress_vbl_this_frame = true; ... }`).
8899        //
8900        // The exact rendering of "read on the same dot as set"
8901        // depends on whether the set callback in tick() fires
8902        // BEFORE the read or AFTER. Since the test ticks the PPU
8903        // to a position FIRST then issues a synchronous read in the
8904        // same test step, the read sees the post-tick state — i.e.
8905        // the read on dot 1 of scanline 241 sees VBL set.
8906        let expected: [ExpectedRow; 10] = [
8907            ExpectedRow {
8908                scanline: 240,
8909                dot: 339,
8910                read_bit7: Some(0),
8911                suppress_set: Some(false),
8912                vblank_after_read: Some(false),
8913            },
8914            ExpectedRow {
8915                scanline: 240,
8916                dot: 340,
8917                read_bit7: Some(0),
8918                suppress_set: Some(false),
8919                vblank_after_read: Some(false),
8920            },
8921            ExpectedRow {
8922                scanline: 241,
8923                dot: 0,
8924                read_bit7: Some(0),
8925                suppress_set: Some(true),
8926                vblank_after_read: Some(false),
8927            },
8928            ExpectedRow {
8929                scanline: 241,
8930                dot: 1,
8931                // VBL is set on this tick BEFORE the read; the read
8932                // returns 1 AND `suppress_vbl_this_frame` is latched
8933                // (RustyNES's `dot <= 1` race window — 2 PPU dots wide).
8934                //
8935                // Session-18 / C1 attempt 16 (PPU-axis, rolled back):
8936                // tightening the predicate to `dot == 0` (matching
8937                // Mesen2 + nesdev wiki) did not flip the failing
8938                // `cpu_interrupts_v2/{2,3,5}` tests at the integration
8939                // layer — the load-bearing axis is the CPU-vs-PPU
8940                // intra-cycle access interleaving, not the suppression
8941                // predicate's literal dot range. Restored 2-dot window
8942                // as the cleaner regression invariant; the unit test
8943                // documents the actual behavior. See
8944                // `docs/audit/session-18-c1-attempt16-ppu-axis-rollback-2026-05-22.md`.
8945                read_bit7: Some(1),
8946                // R2 (mc-r1-substrate): the dot==0 race window (Mesen2
8947                // `UpdateStatusFlag:590`) makes the dot-1 read a normal
8948                // post-set read — no suppression. Default (2-dot window):
8949                // suppression latches at dot 1 too.
8950                suppress_set: Some(false),
8951                vblank_after_read: Some(false),
8952            },
8953            ExpectedRow {
8954                scanline: 241,
8955                dot: 2,
8956                read_bit7: Some(1),
8957                suppress_set: Some(false),
8958                vblank_after_read: Some(false),
8959            },
8960            ExpectedRow {
8961                scanline: 241,
8962                dot: 3,
8963                read_bit7: Some(1),
8964                suppress_set: Some(false),
8965                vblank_after_read: Some(false),
8966            },
8967            ExpectedRow {
8968                scanline: 241,
8969                dot: 4,
8970                read_bit7: Some(1),
8971                suppress_set: Some(false),
8972                vblank_after_read: Some(false),
8973            },
8974            ExpectedRow {
8975                scanline: 241,
8976                dot: 5,
8977                read_bit7: Some(1),
8978                suppress_set: Some(false),
8979                vblank_after_read: Some(false),
8980            },
8981            ExpectedRow {
8982                scanline: 242,
8983                dot: 0,
8984                read_bit7: Some(1),
8985                suppress_set: Some(false),
8986                vblank_after_read: Some(false),
8987            },
8988            ExpectedRow {
8989                scanline: 242,
8990                dot: 1,
8991                read_bit7: Some(1),
8992                suppress_set: Some(false),
8993                vblank_after_read: Some(false),
8994            },
8995        ];
8996
8997        // Per-row capture for human inspection.
8998        #[derive(Debug)]
8999        struct ObservedRow {
9000            scanline: i16,
9001            dot: u16,
9002            read_value: u8,
9003            read_bit7: u8,
9004            suppress_set_after: bool,
9005            status_vblank_after: bool,
9006        }
9007        let mut observed = Vec::<ObservedRow>::new();
9008
9009        for row in &expected {
9010            // Build a fresh PPU and tick it to the target (scanline, dot).
9011            // Strategy: tick UNTIL we land on the target. Each `tick`
9012            // call calls `advance_dot()` first, so the post-tick state
9013            // is (scanline + 1, dot=1 wraparound) etc. Hence we tick
9014            // until p.scanline()/p.dot() match.
9015            //
9016            // Disable rendering so we don't trigger A12 emissions,
9017            // sprite eval, etc. — keeps the test focused on VBL +
9018            // $2002.
9019            let (mut p, mut b) = fresh_ppu();
9020            // No PPUMASK render bits. No PPUCTRL bits (so NMI off).
9021            // post_reset_mask_remaining = 0 already (fresh_ppu sets it).
9022            // Tick to the target. Loop bound is one full NTSC frame plus
9023            // safety margin (see `VBL_SWEEP_MAX_TICKS` above).
9024            let mut ticks = 0u32;
9025            while !(p.scanline == row.scanline && p.dot == row.dot) {
9026                p.tick(&mut b);
9027                ticks += 1;
9028                assert!(
9029                    ticks < VBL_SWEEP_MAX_TICKS,
9030                    "could not reach (scanline={}, dot={}) within one frame; \
9031                     loop bug or scheduler change",
9032                    row.scanline,
9033                    row.dot,
9034                );
9035            }
9036
9037            // Issue the $2002 read.
9038            let read_value = p.cpu_read_register(2, &mut b);
9039            let read_bit7 = (read_value >> 7) & 1;
9040            let suppress_set_after = p.suppress_vbl_this_frame;
9041            let status_vblank_after = p.status.contains(PpuStatus::VBLANK);
9042
9043            observed.push(ObservedRow {
9044                scanline: row.scanline,
9045                dot: row.dot,
9046                read_value,
9047                read_bit7,
9048                suppress_set_after,
9049                status_vblank_after,
9050            });
9051        }
9052
9053        // Print the table for human inspection (only visible with
9054        // --nocapture).
9055        eprintln!();
9056        eprintln!("=== $2002 VBL race-window sweep ===");
9057        eprintln!(
9058            "{:>3} {:>3}  {:>8} {:>7} {:>11} {:>14}",
9059            "sl", "dot", "read", "bit7", "suppress?", "PPU.VBLANK?",
9060        );
9061        for o in &observed {
9062            eprintln!(
9063                "{:>3} {:>3}  0x{:02X}     {:>5}    {:>9}    {:>10}",
9064                o.scanline,
9065                o.dot,
9066                o.read_value,
9067                o.read_bit7,
9068                o.suppress_set_after,
9069                o.status_vblank_after,
9070            );
9071        }
9072        eprintln!();
9073
9074        // Assert the spec — but only on rows where `expected` carries
9075        // a concrete claim. Rows with `None` are recording-only.
9076        let mut failures = Vec::<String>::new();
9077        for (i, row) in expected.iter().enumerate() {
9078            let obs = &observed[i];
9079            if let Some(want) = row.read_bit7
9080                && obs.read_bit7 != want
9081            {
9082                failures.push(format!(
9083                    "(sl={}, dot={}): expected read bit7 = {}, got {}",
9084                    row.scanline, row.dot, want, obs.read_bit7,
9085                ));
9086            }
9087            if let Some(want) = row.suppress_set
9088                && obs.suppress_set_after != want
9089            {
9090                failures.push(format!(
9091                    "(sl={}, dot={}): expected suppress_vbl = {}, got {}",
9092                    row.scanline, row.dot, want, obs.suppress_set_after,
9093                ));
9094            }
9095            if let Some(want) = row.vblank_after_read
9096                && obs.status_vblank_after != want
9097            {
9098                failures.push(format!(
9099                    "(sl={}, dot={}): expected status.VBLANK after read = {}, got {}",
9100                    row.scanline, row.dot, want, obs.status_vblank_after,
9101                ));
9102            }
9103        }
9104
9105        assert!(
9106            failures.is_empty(),
9107            "$2002 race-window sweep mismatches vs. nesdev wiki spec:\n  {}",
9108            failures.join("\n  "),
9109        );
9110    }
9111
9112    // -------------------------------------------------------------------
9113    // v1.3.x left-edge regression: BG attribute (palette) shift register
9114    // must stay in lockstep with the BG pattern shift registers through
9115    // the dots 321-336 pre-fetch boundary.
9116    //
9117    // 086ce4d moved the BG pattern pipeline to the Mesen2 cycle-9 reload +
9118    // post-emit shift model and added an explicit `<<= 8` at pre-fetch
9119    // dots 328/336 for the 16-bit pattern shifters, but left the 8-bit
9120    // `at_shift` + 1-bit `at_feed` attribute model untouched. The two
9121    // pipelines then advanced at different rates across the pre-fetch
9122    // region, so the palette (attribute) bits drifted one tile out of
9123    // phase with the pattern bits in the leftmost columns — the source of
9124    // the "green tint / garbage palette in the left 1-2 columns while
9125    // scrolling" regression. The fix makes the attribute shifters 16-bit
9126    // and shift them in lockstep with the pattern shifters.
9127    // -------------------------------------------------------------------
9128
9129    /// Render one visible scanline with a SOLID pattern everywhere
9130    /// (pattern value 1 in every tile) but a per-tile-group ATTRIBUTE
9131    /// boundary, then return the (pattern, palette) the PPU emitted at
9132    /// each of the first 24 columns. Because the pattern value is the
9133    /// same everywhere, any column-to-column change is purely an
9134    /// attribute (palette) change — which is exactly what the AT shift
9135    /// register controls. Misalignment between the pattern and attribute
9136    /// pipelines therefore shows up as the palette boundary landing on
9137    /// the wrong column.
9138    ///
9139    /// Returns a Vec of `(palette_value)` per column 0..24 of the target
9140    /// scanline (pattern value is always 1, verified internally).
9141    fn diag_attr_palette_per_column(fine_x: u8, coarse_x: u16) -> alloc::vec::Vec<u8> {
9142        use alloc::vec::Vec;
9143        let target_line: usize = 5;
9144        let (mut p, mut b) = fresh_ppu();
9145
9146        // Single solid tile: tile 1 = pattern value 1 on all rows.
9147        for row in 0..8u16 {
9148            b.chr[(0x0010 + row) as usize] = 0xFF; // lo plane all set
9149            b.chr[(0x0018 + row) as usize] = 0x00; // hi plane clear -> value 1
9150        }
9151
9152        // Nametable 0: every tile = tile 1 (solid). Attribute table sets
9153        // a palette boundary: tile-column groups 0-1 use palette 1,
9154        // everything else uses palette 0. Each attribute byte covers a
9155        // 4x4-tile (32x32px) region split into four 2x2-tile quadrants.
9156        // We set the top-left quadrant of attribute byte 0 to palette 1.
9157        for off in 0..0x03C0u16 {
9158            p.ciram[off as usize] = 0x01; // tile index 1 everywhere
9159        }
9160        // Attribute table starts at $23C0 -> CIRAM offset 0x03C0.
9161        // Byte 0 covers tile columns 0-3, rows 0-3. Bits 1-0 = top-left
9162        // quadrant (tile cols 0-1, rows 0-1); bits 3-2 = top-right
9163        // quadrant (tile cols 2-3, rows 0-1). Set TL=palette 1, TR=
9164        // palette 2, the rest palette 0. This puts an attribute boundary
9165        // at every 16px (tile-pair) step so coarse-X scroll moves the
9166        // boundary across the pre-fetch-fed leftmost tile — the exact
9167        // condition that exposed the 086ce4d AT lockstep regression.
9168        p.ciram[0x03C0] = 0b00_00_10_01; // TL=pal1, TR=pal2.
9169        // The target scanline is row 5 -> tile row 0 -> top quadrants.
9170
9171        // Palettes: pattern value 1...
9172        //   palette 0 -> $3F01
9173        //   palette 1 -> $3F05
9174        //   palette 2 -> $3F09
9175        p.palette_ram[palette_index(0x3F00)] = 0x0F; // universal
9176        p.palette_ram[palette_index(0x3F01)] = 0x30; // pal0 value1 = white
9177        p.palette_ram[palette_index(0x3F05)] = 0x16; // pal1 value1 = red
9178        p.palette_ram[palette_index(0x3F09)] = 0x2A; // pal2 value1 = green
9179
9180        // No sprites.
9181        for i in 0..256 {
9182            p.oam[i] = 0xF0;
9183        }
9184
9185        p.ctrl = PpuCtrl::empty();
9186        p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_BG_LEFT;
9187
9188        // Scroll: coarse-X into t bits 0-4, fine-x into p.x.
9189        p.t = coarse_x & 0x1F;
9190        p.v = 0;
9191        p.x = fine_x;
9192
9193        p.scanline = p.region.prerender_line();
9194        p.dot = 0;
9195        p.last_a12_level = false;
9196
9197        for _ in 0..(341 * (target_line + 2)) {
9198            p.tick(&mut b);
9199        }
9200
9201        let line = target_line;
9202        let pal0 = crate::palette::nes_color_to_rgba(0x30);
9203        let pal1 = crate::palette::nes_color_to_rgba(0x16);
9204        let pal2 = crate::palette::nes_color_to_rgba(0x2A);
9205        let universal = crate::palette::nes_color_to_rgba(0x0F);
9206        let mut out = Vec::with_capacity(24);
9207        for x in 0..24usize {
9208            let off = (line * 256 + x) * 4;
9209            let px = [
9210                p.framebuffer[off],
9211                p.framebuffer[off + 1],
9212                p.framebuffer[off + 2],
9213                p.framebuffer[off + 3],
9214            ];
9215            // Map color back to palette index: 0/1/2 = palette, 254 =
9216            // universal, 255 = unexpected.
9217            let v = if px == pal0 {
9218                0u8
9219            } else if px == pal1 {
9220                1u8
9221            } else if px == pal2 {
9222                2u8
9223            } else if px == universal {
9224                254u8
9225            } else {
9226                255u8
9227            };
9228            out.push(v);
9229        }
9230        out
9231    }
9232
9233    /// The expected palette index for screen column `x` given a scroll of
9234    /// `coarse_x` tiles + `fine_x` pixels. Tile column C maps to: 0-1 ->
9235    /// palette 1, 2-3 -> palette 2, 4+ -> palette 0. With a total left
9236    /// shift of `coarse_x*8 + fine_x` pixels, screen column `x`
9237    /// corresponds to source pixel `x + coarse_x*8 + fine_x`, whose tile
9238    /// column is that pixel / 8.
9239    fn expected_palette(x: usize, fine_x: u8, coarse_x: u16) -> u8 {
9240        let src_pixel = x + (coarse_x as usize) * 8 + fine_x as usize;
9241        let tile_col = src_pixel / 8;
9242        match tile_col {
9243            0 | 1 => 1,
9244            2 | 3 => 2,
9245            _ => 0,
9246        }
9247    }
9248
9249    /// With NO scroll the palette-1 region must cover exactly tile columns
9250    /// 0-1 (screen columns 0-15), palette-2 tile columns 2-3 (16-31), and
9251    /// palette-0 beyond. Visible-region pipeline only; must hold both
9252    /// before and after the fix.
9253    #[test]
9254    fn bg_attribute_boundary_no_scroll() {
9255        let cols = diag_attr_palette_per_column(0, 0);
9256        for (x, &v) in cols.iter().enumerate() {
9257            let want = expected_palette(x, 0, 0);
9258            assert_eq!(
9259                v, want,
9260                "col {x}: expected palette {want}, got {v} (full: {cols:?})"
9261            );
9262        }
9263    }
9264
9265    /// The palette boundary must stay glued to the pattern across BOTH
9266    /// fine-X and coarse-X scroll. This is the case the 086ce4d AT-register
9267    /// regression broke: the pre-fetch `<<= 8` (added for the 16-bit
9268    /// pattern shifters) was not applied to the 8-bit attribute model, so
9269    /// the palette drifted one tile relative to the pattern across the
9270    /// dots 321-336 pre-fetch boundary — wrong palette in the leftmost
9271    /// tile column (screen columns 0-7). With the 16-bit AT shifters the
9272    /// boundary tracks the pattern exactly at every scroll value.
9273    ///
9274    /// Empirically: on the pre-fix (HEAD) tree the `coarse_x` cases below
9275    /// mis-paint screen columns 0-7; on the fixed tree every column
9276    /// matches `expected_palette`.
9277    #[test]
9278    fn bg_attribute_boundary_tracks_pattern_under_scroll() {
9279        for coarse_x in 0..6u16 {
9280            for fine_x in 0..8u8 {
9281                let cols = diag_attr_palette_per_column(fine_x, coarse_x);
9282                for (x, &v) in cols.iter().enumerate() {
9283                    let want = expected_palette(x, fine_x, coarse_x);
9284                    assert_eq!(
9285                        v, want,
9286                        "coarse_x={coarse_x} fine_x={fine_x} col {x}: \
9287                         expected palette {want}, got {v}\nfull: {cols:?}\n\
9288                         (palette boundary must stay glued to the pattern; \
9289                         a mismatch in columns 0-7 is the 086ce4d \
9290                         AT-register lockstep regression)"
9291                    );
9292                }
9293            }
9294        }
9295    }
9296
9297    /// v2.7.6 (core audit IMP-06) — the fast render path's history invariant
9298    /// is checked, not re-imposed.
9299    ///
9300    /// The fast body used to write `true` into the three rendering-history
9301    /// fields, which the guard in `tick` already requires. Those writes were
9302    /// replaced by a `debug_assert!`. This test enters the fast body with a
9303    /// history that lags the mask (the first dot after an enable) and requires
9304    /// the assertion to fire. It has to be a direct unit test: no ROM in the
9305    /// corpus can reach this state, because the guard's `mask_write_delay`
9306    /// term alone keeps such dots off the fast path. Deleting every history
9307    /// term from the guard left `fast_dotloop_diff` green, which is why the
9308    /// assertion is pinned here rather than through it.
9309    #[test]
9310    #[cfg(debug_assertions)]
9311    #[should_panic(
9312        expected = "fast render path entered without a stably-enabled rendering history"
9313    )]
9314    fn fast_render_path_asserts_its_rendering_history() {
9315        let (mut p, mut bus) = fresh_ppu();
9316        p.dot = 1;
9317        p.prev_rendering_enabled = false;
9318        p.rendering_enabled_delayed = true;
9319        p.rendering_enabled_delayed2 = true;
9320        p.tick_visible_render_fast(&mut bus);
9321    }
9322}