rustynes_ppu/ppu.rs
1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: this PPU contains code derived from Mesen2 (GPL-3.0-or-later): the sprite-evaluation FSM and OAM-data-bus model, `Core/NES/NesPpu.cpp` (`ProcessSpriteEvaluation` / `ReadSpriteRam`), and the optional OAM-decay model (`ReadSpriteRam` / `WriteSpriteRam`, `OamDecayCycleCount`; added v2.1.4, disclosed v2.9.4); it also incorporates models ported from TriCNES (MIT) — the ALE / octal-latch address-multiplex and the OAM-corruption behavior. See docs/originality-and-provenance.md (Section 1)
4// and NOTICE for the complete, audited derivation record.
5//! 2C02 PPU core: state, register surface, scanline counter, NMI signaling.
6//!
7//! See `docs/ppu-2c02.md`. Background and sprite *rendering* (per-dot tile
8//! fetch, shift registers, sprite evaluation, sprite-zero hit) is plumbed
9//! through this struct but the visible-pixel output path is filled in by
10//! Sprints 2-2 and 2-3 — the surface and scanline FSM here is what
11//! Sprint 2-1 delivers.
12
13use crate::bus::{BgSplitState, ExAttribute, PpuBus};
14use crate::palette::{build_rgba_lut, build_rgba_lut_from_base};
15use crate::registers::{PpuCtrl, PpuMask, PpuStatus};
16use alloc::boxed::Box;
17use alloc::vec;
18
19/// Visible screen width in pixels, and the stride of every per-pixel buffer the
20/// PPU exposes.
21///
22/// v2.3.8 — the single definition. `provenance::SCREEN_W` was a second copy of
23/// this number in a `debug-hooks`-gated module, which made the width
24/// unreachable from ungated code and invited a third copy rather than a
25/// dependency. It now aliases this.
26pub const SCREEN_WIDTH: usize = 256;
27
28/// Visible screen height in pixels. Companion to [`SCREEN_WIDTH`].
29pub const SCREEN_HEIGHT: usize = 240;
30
31/// RGBA8 framebuffer length in bytes (256 × 240 × 4).
32pub const FRAMEBUFFER_LEN: usize = SCREEN_WIDTH * SCREEN_HEIGHT * 4;
33
34/// Visible pixel count (256 × 240) — length of the parallel
35/// [`Ppu::index_framebuffer`] (one `u16` per pixel).
36pub const FRAMEBUFFER_PIXELS: usize = SCREEN_WIDTH * SCREEN_HEIGHT;
37
38/// v3.1.0 (`T-SPRITE-LIMIT`) — the most sprites the "disable sprite limit"
39/// option can add to one scanline: all 64 OAM entries minus the eight the
40/// hardware draws.
41pub const MAX_EXTRA_SPRITES: usize = 64 - 8;
42
43/// v1.2.0 beta.2 (Workstream C3) — per-pixel HD-pack tile-source record.
44///
45/// One entry per visible pixel (parallel to [`Ppu::index_framebuffer`]),
46/// populated in the pixel-emit path only when the `hd-pack` cargo feature is
47/// enabled. It records the **identity of the CHR tile** that produced the
48/// pixel — the 16-byte pattern-table tile base address (in PPU `$0000..=$1FFF`
49/// pattern space, fine-Y masked off), the 2-bit attribute/sprite palette, the
50/// sprite flip flags, and whether the source was a sprite or the background.
51///
52/// It is pure output telemetry: it mirrors data the renderer already computed,
53/// reads no new VRAM, and changes no emulation state — so it is byte-identical
54/// with the feature on or off and is never serialized into a save-state. The
55/// frontend's Mesen-style HD-pack loader groups these by 8×8 screen cell, hashes
56/// the referenced CHR bytes, and substitutes hi-res replacement tiles at blit
57/// time. See `docs/ppu-2c02.md` §HD-pack tile-source export.
58#[cfg(feature = "hd-pack")]
59#[derive(Clone, Copy, Debug, Eq, PartialEq)]
60pub struct HdTileSource {
61 /// 16-byte CHR tile base address in pattern space (`$0000..=$1FF0`, low
62 /// nibble always 0). `tile = (addr >> 4) & 0xFF`, `table = addr & 0x1000`.
63 /// `0xFFFF` marks a transparent / universal-background pixel (no tile).
64 pub chr_addr: u16,
65 /// Final palette group: BG attribute (0..=3) or sprite palette (0..=3).
66 pub palette: u8,
67 /// `true` when the pixel came from a sprite (vs. the background).
68 pub is_sprite: bool,
69 /// Sprite horizontal flip (always `false` for BG pixels).
70 pub flip_h: bool,
71 /// Sprite vertical flip (always `false` for BG pixels).
72 pub flip_v: bool,
73 /// Mesen `PaletteColors`: the tile's active palette packed as the HD-pack
74 /// tile-identity key expects. BG = `pr[base+3] | pr[base+2]<<8 |
75 /// pr[base+1]<<16 | pr[0]<<24`; sprite = `0xFF000000 | pr[base+3] |
76 /// pr[base+2]<<8 | pr[base+1]<<16` (top byte `0xFF` = the sprite/BG
77 /// discriminator). Part of the CHR-RAM/CHR-ROM tile key (HdNesPpu.h:119/167).
78 pub palette_colors: u32,
79 /// Which texel COLUMN (0..=7) of the 8x8 tile this screen pixel samples —
80 /// Mesen `OffsetX` (HdNesPpu.h:172). For BG it folds in fine-X scroll
81 /// (`(fineX + (pixel_x & 7)) & 7`); for a sprite it is the column within the
82 /// sprite tile. The HD compositor samples the replacement at this column so
83 /// the high-def tile tracks the scrolled / sprite position pixel-for-pixel
84 /// (flips are applied at sample time). Output-only.
85 pub offset_x: u8,
86 /// Which texel ROW (0..=7) of the 8x8 tile this screen pixel samples — Mesen
87 /// `OffsetY`. BG = fine-Y; sprite = the row within the sprite tile.
88 pub offset_y: u8,
89 /// Mesen CHR-ROM `TileIndex`: the ABSOLUTE post-banking CHR-ROM tile number
90 /// (`chr_phys(addr) / 16`) for a CHR-ROM cart, or [`HD_CHR_RAM`] when CHR is
91 /// RAM (the tile is content-hashed instead). The HD-pack key uses
92 /// `TileIndex ^ PaletteColors` for CHR-ROM and `CalculateHash(palette ++
93 /// data)` for CHR-RAM (Mesen `HdTileKey`). Captured at fetch (the only point
94 /// with mapper access). Output-only.
95 pub chr_tile_index: u32,
96 /// v1.8.9 — the `$2001` grayscale + emphasis bits at this pixel
97 /// (`mask.bits() & 0xE1`: bit 0 = grayscale, bits 5-7 = R/G/B emphasis). The
98 /// HD compositor re-applies them to the replacement texel (Mesen
99 /// `ProcessGrayscaleAndEmphasis`) so HD tiles track grayscale / emphasis fades
100 /// like the base frame, which already has them baked in. Output-only.
101 pub color_mask: u8,
102 /// v1.8.9 — every opaque sprite covering this pixel (up to 4), front-to-back,
103 /// for Mesen `spriteAtPosition` / `spriteNearby` (which match ANY covering
104 /// sprite, including ones a higher-priority BG occludes). `sprites[0]` is the
105 /// front-most. The existing `is_sprite` + tile fields still describe the
106 /// VISIBLE pixel (the winning layer); these add the hidden layers. Output-only.
107 pub sprites: [HdSprite; 4],
108 /// Number of valid entries in [`Self::sprites`] (`0..=4`).
109 pub sprite_count: u8,
110}
111
112/// One sprite covering a pixel, for the HD-pack multi-sprite conditions
113/// (`spriteAtPosition` / `spriteNearby`). Carries just the identity those
114/// conditions match on. Output-only telemetry.
115#[cfg(feature = "hd-pack")]
116#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
117pub struct HdSprite {
118 /// Absolute CHR-ROM tile index (`chr_phys/16`), or [`HD_CHR_RAM`] for CHR-RAM.
119 pub chr_tile_index: u32,
120 /// The sprite's packed `PaletteColors` (Mesen key form).
121 pub palette_colors: u32,
122}
123
124/// `chr_tile_index` sentinel meaning "CHR is RAM" — the tile is keyed by its 16
125/// CHR bytes (content) rather than by an absolute CHR-ROM tile index.
126#[cfg(feature = "hd-pack")]
127pub const HD_CHR_RAM: u32 = u32::MAX;
128
129#[cfg(feature = "hd-pack")]
130impl Default for HdTileSource {
131 /// A blank record: no tile, and the CHR-RAM sentinel (so an unwritten /
132 /// default record is content-keyed, never mistaken for CHR-ROM tile 0).
133 fn default() -> Self {
134 Self {
135 chr_addr: 0,
136 palette: 0,
137 is_sprite: false,
138 flip_h: false,
139 flip_v: false,
140 palette_colors: 0,
141 offset_x: 0,
142 offset_y: 0,
143 chr_tile_index: HD_CHR_RAM,
144 color_mask: 0,
145 sprites: [HdSprite::default(); 4],
146 sprite_count: 0,
147 }
148 }
149}
150
151/// Sentinel `chr_addr` for a transparent / universal-background HD-pack pixel.
152#[cfg(feature = "hd-pack")]
153pub const HD_TILE_NONE: u16 = 0xFFFF;
154
155/// Diagnostic: capture the (frame, scanline, dot, mask) of `$2007` reads to
156/// pin where the `$2007 Stress` test's per-dot reads land vs the visible
157/// scanline they target. Gated; default build unaffected.
158pub mod read2007_diag {
159 use core::sync::atomic::AtomicU32;
160 /// Next free slot (count of captured `$2007` reads).
161 pub static IDX: AtomicU32 = AtomicU32::new(0);
162 /// Packed: `((scanline+1)<<18) | (dot<<5) | (rendering_enabled<<1) | is_render`.
163 pub static LOG: [AtomicU32; 1024] = [const { AtomicU32::new(0) }; 1024];
164 /// Tunable PPU-dot countdown: `$2007` read-during-rendering to reload.
165 ///
166 /// The `PPUDATA` state machine reloads `data_buffer` from the fetch cadence
167 /// (`TriCNES` `PPU_DATA_StateMachine` latch cascade: read-end -> ALE +2 ->
168 /// reload +4 dots; with R1's fixed CPU<->PPU mod-4 phase the landing is a
169 /// constant offset from the register-read sample point). 0 = immediate.
170 /// Default 5 = the empirical winner (W2 sweep 0-12: 5 -> 170/170 stable
171 /// reads on the `$2007 Stress` answer key; 6/7 -> 169; everything else
172 /// far below). Env knob `RUSTYNES_2007_DELAY` (wired in the diag bins).
173 pub static RENDER_BUFFER_DOT_DELAY: AtomicU32 = AtomicU32::new(5);
174 /// Sub-knob: defer the `$2007` v-glitch increment to the `TStep` dot.
175 ///
176 /// The `TStep` is the SAME dot as the buffer reload — `TriCNES`
177 /// `PPU_DATA_StateMachine_Half`: `PPU_2007_TStep = TStep_Latch || PD_RB`
178 /// — instead of read time. With the immediate increment, every fetch in
179 /// the read-to-reload window uses the post-glitch `v` (coarse-x +1 -> NT
180 /// byte = tile+1; fine-y +1 -> PT row+1) — exactly the per-index mismatch
181 /// signature the W2 baseline measured. 1 = deferred (default), 0 =
182 /// immediate (legacy). Env knob `RUSTYNES_2007_VINC` (wired in the diag
183 /// bins).
184 pub static RENDER_BUFFER_DEFER_V_INC: AtomicU32 = AtomicU32::new(1);
185}
186
187/// v2.0.3 (ADR 0030) octal-latch calibration tracer.
188///
189/// Entirely a diagnostic: a lock-free ring of packed per-event records captured
190/// only when `ENABLE` is set (via the env knob wired in the test harness). It
191/// costs nothing in an untraced run — each `push` call site is a single relaxed
192/// atomic load + early-return branch, and every call site sits in a cold
193/// corruption branch (never the steady-state per-dot path). Records the
194/// corruption-relevant events (`$2006`/`$2007` during render, hybrid/stale
195/// splices) on scanlines 2-5 so a run can be cross-diffed against the `TriCNES`
196/// per-dot bus sequence.
197///
198/// Gated behind the default-off `ppu-octal-trace` dev feature. Behavior never
199/// depends on it — the 2-cycle-ALE fetch model (now the only PPU fetch path, ADR
200/// 0030) drives its `push` call sites unconditionally, but with the feature off
201/// `push` is a zero-cost no-op ([`stub`](self)) and the ring's 64-bit-atomic
202/// storage does not exist, so the shipped hot path is untouched and the
203/// `#![no_std]` chip stack (whose `thumbv7em` target lacks 64-bit atomics) still
204/// builds. Enable with `--features ppu-octal-trace` to capture a trace.
205#[cfg(feature = "ppu-octal-trace")]
206pub mod octal_trace {
207 use core::sync::atomic::{AtomicU32, AtomicU64};
208 /// 1 = capture enabled. Off by default; an untraced flag-on run pays only a
209 /// single relaxed atomic load + branch per `push` call site.
210 pub static ENABLE: AtomicU32 = AtomicU32::new(0);
211 /// Next free slot (saturating; stops at `LOG.len()`).
212 pub static IDX: AtomicU32 = AtomicU32::new(0);
213 /// Packed record: `(kind<<58) | (frame<<44) | (scanline<<32) | (dot<<20) | value`
214 /// where `value` is event-specific (a 20-bit address / latch payload).
215 pub static LOG: [AtomicU64; 4096] = [const { AtomicU64::new(0) }; 4096];
216
217 /// Event kind: `$2006` second write during rendering (value = new `v`).
218 pub const K_W2006: u64 = 1;
219 /// Event kind: `$2007` read during rendering (value = `v`).
220 pub const K_R2007: u64 = 2;
221 /// Event kind: hybrid nametable splice fired (value = effective address).
222 pub const K_HYBRID: u64 = 3;
223 /// Event kind: stale-latch pattern splice fired (value = effective address).
224 pub const K_STALE: u64 = 4;
225 /// Event kind: `$2007` state-machine countdown landed (value = data byte).
226 pub const K_SMLAND: u64 = 5;
227
228 /// Push one record (no-op unless `ENABLE` and slots remain).
229 #[allow(clippy::cast_sign_loss)] // scanline filtered to 2..=5 (always >= 0)
230 pub fn push(kind: u64, frame: u64, scanline: i16, dot: u16, value: u32) {
231 if ENABLE.load(core::sync::atomic::Ordering::Relaxed) == 0 {
232 return;
233 }
234 // Restrict to the corruption-relevant visible scanlines (ALE+Read is
235 // scanline 3, Hybrid scanline 4) to keep the ring from overflowing on
236 // the many unrelated `$2006`/`$2007`-during-render writes elsewhere.
237 if !(2..=5).contains(&scanline) {
238 return;
239 }
240 let i = IDX.fetch_add(1, core::sync::atomic::Ordering::Relaxed) as usize;
241 if i >= LOG.len() {
242 return;
243 }
244 let sl = (u64::from((scanline & 0x0FFF) as u16)) << 32;
245 let packed = (kind << 58)
246 | ((frame & 0x3FFF) << 44)
247 | sl
248 | ((u64::from(dot) & 0xFFF) << 20)
249 | u64::from(value & 0xF_FFFF);
250 LOG[i].store(packed, core::sync::atomic::Ordering::Relaxed);
251 }
252}
253
254/// Zero-cost no-op stand-in for the octal-latch tracer.
255///
256/// Compiled when the `ppu-octal-trace` dev feature is off (the default, and the
257/// only config the `#![no_std]` chip stack builds — the full tracer's ring needs
258/// 64-bit atomics the `thumbv7em` target lacks). Exposes the same `push`
259/// signature and `K_*` event-kind constants the 2-cycle-ALE fetch path
260/// references, so those call sites stay unconditional; `push` here is an empty
261/// body the optimizer removes entirely. Behavior is thus byte-identical to the
262/// full tracer with capture disabled — the tracer only ever observes, never
263/// influences, emulation.
264#[cfg(not(feature = "ppu-octal-trace"))]
265pub mod octal_trace {
266 /// Event kind: `$2006` second write during rendering.
267 pub const K_W2006: u64 = 1;
268 /// Event kind: `$2007` read during rendering.
269 pub const K_R2007: u64 = 2;
270 /// Event kind: hybrid nametable splice fired.
271 pub const K_HYBRID: u64 = 3;
272 /// Event kind: stale-latch pattern splice fired.
273 pub const K_STALE: u64 = 4;
274 /// Event kind: `$2007` state-machine countdown landed.
275 pub const K_SMLAND: u64 = 5;
276
277 /// No-op (the `ppu-octal-trace` feature is off). Compiles to nothing.
278 #[inline(always)]
279 pub const fn push(_kind: u64, _frame: u64, _scanline: i16, _dot: u16, _value: u32) {}
280}
281
282/// v2.0.3 (ADR 0030, Option 1) — the delayed-`CopyV` countdown length in PPU
283/// dots (`TriCNES` `PPU_Update2006Delay`, `Emulator.cs:9837-9843`, which is 4
284/// for three of the four CPU/PPU sub-cycle alignments and 5 for the fourth).
285/// `RustyNES`'s lockstep bus applies the `$2006` write at the start of a CPU
286/// cycle; the corrupted nametable read is the phase-1 dot of the fetch group one
287/// coarse-X past the write. Empirically calibrated against the `TriCNES` per-dot
288/// bus trace so the countdown lands on that read after exactly one `inc_hori_v`
289/// and one phase-0 NT ALE (which loads the one-tile-ahead `$19` low byte). See
290/// the v2.0.3 campaign plan.
291const COPY_V_DELAY: u8 = 4;
292
293/// v2.1.4 F2.3 — optional OAM decay threshold, in **CPU cycles**.
294///
295/// Provenance: the OAM-decay model (this constant, the per-row timestamps, the
296/// refresh-on-read/write rule and the decayed-byte pattern in
297/// `oam_decay_on_read`) is **derived from Mesen2's `NesPpu.cpp`**
298/// (`ReadSpriteRam` / `WriteSpriteRam`, `OamDecayCycleCount`),
299/// GPL-3.0-or-later. It was written that way at v2.1.4 (#265, whose commit
300/// says so) and recorded in this file's header, NOTICE and
301/// docs/originality-and-provenance.md (Section 1) only at v2.9.4.
302///
303/// The 2C02's Object Attribute Memory is dynamic RAM: each row is implicitly
304/// refreshed every time sprite evaluation (or a `$2004` access) reads it during
305/// rendering, but with rendering disabled long enough the un-refreshed rows lose
306/// their charge and decay to a fixed garbage pattern. Mesen2 models this as a
307/// per-8-byte-row CPU-cycle timestamp with a 3000-cycle refresh window
308/// (`NesPpu::OamDecayCycleCount`, `Core/NES/NesPpu.cpp`); a read/write that lands
309/// within 3000 CPU cycles of the row's last touch refreshes it, otherwise the row
310/// has decayed. This value mirrors Mesen2's constant exactly so the two agree on
311/// when a row is considered stale.
312///
313/// This whole model is **off by default** (`Ppu::oam_decay_enabled == false`) and
314/// **NTSC/Dendy-only** — on PAL the far more frequent refresh cadence masks decay
315/// entirely, so the feature is never applied there. With it off, no OAM access
316/// consults the decay state and the PPU is byte-identical to a build that never
317/// had the field. See `docs/ppu-2c02.md` (§OAM decay).
318const OAM_DECAY_CPU_CYCLES: u64 = 3000;
319
320/// Region governs the size of the post-render-to-pre-render scanline span.
321#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash)]
322pub enum PpuRegion {
323 /// NTSC (and Famicom). 262 scanlines per frame, pre-render = scanline 261.
324 Ntsc,
325 /// PAL. 312 scanlines per frame, pre-render = scanline 311.
326 Pal,
327 /// Dendy (Russian PAL famiclone). 312 scanlines, but VBL starts at 291.
328 Dendy,
329}
330
331impl PpuRegion {
332 /// Pre-render scanline number.
333 #[must_use]
334 pub const fn prerender_line(self) -> i16 {
335 match self {
336 Self::Ntsc => 261,
337 Self::Pal | Self::Dendy => 311,
338 }
339 }
340
341 /// Last visible scanline (always 239).
342 #[must_use]
343 pub const fn last_visible_line(self) -> i16 {
344 239
345 }
346
347 /// Scanline at which V-blank starts (and `PPUSTATUS.VBLANK` is set on dot 1).
348 #[must_use]
349 pub const fn vblank_start_line(self) -> i16 {
350 match self {
351 Self::Ntsc | Self::Pal => 241,
352 Self::Dendy => 291,
353 }
354 }
355
356 /// Number of CPU cycles `$2000`/`$2001`/`$2005`/`$2006` writes are
357 /// ignored after a power-on / reset. Per nesdev wiki:
358 /// NTSC ≈ 29,658; PAL ≈ 33,132.
359 #[must_use]
360 pub const fn post_reset_mask_cycles(self) -> u32 {
361 match self {
362 Self::Ntsc => 29_658,
363 Self::Pal | Self::Dendy => 33_132,
364 }
365 }
366}
367
368/// v2.1.7 P5 — selectable 2C02 die revision, gating revision-dependent quirks.
369///
370/// Additive and **default-off**: the [`Default`] ([`Self::Rp2c02H`]) preserves
371/// `RustyNES`'s established behavior byte-for-byte, so `AccuracyCoin`, the
372/// commercial oracle, and the visual / audio regression suites are unaffected at
373/// the default. Only the opt-in [`Self::Rp2c02G`] selection changes any emulated
374/// behavior (see below).
375///
376/// Real RP2C02 dies shipped across several letter revisions. The one behavioral
377/// difference `RustyNES` currently models per-revision is the **OAMADDR
378/// (`$2003`) write-during-rendering OAM corruption** glitch: writing `$2003`
379/// while rendering is enabled on a visible / pre-render scanline copies one
380/// 8-byte OAM "row" from row 0 over the row the write's high bits target, on the
381/// next rendered dot (the same `CorruptOAM` mechanism the rendering-disable
382/// model uses; see `Ppu::process_oam_corruption`). A handful of titles —
383/// notably *Huge Insect* — trip it. It is **not** enabled on the default
384/// revision.
385///
386/// Honesty note (see `docs/accuracy-ledger.md`): the exact mapping of the
387/// `$2003` corruption onto specific 2C02 letter revisions is not firmly
388/// established in the public literature, and the precise per-title byte output
389/// of the glitch is not independently oracle-verified in this cut. `RustyNES`
390/// therefore offers the model as an opt-in approximation keyed to a single
391/// "earlier revision" selection ([`Self::Rp2c02G`]) rather than claiming exact
392/// silicon-revision fidelity. This is config, **not** save-state: like
393/// [`PpuRegion`] it is re-applied on load and is not part of the snapshot.
394#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash, Default)]
395pub enum PpuRevision {
396 /// Default. Later RP2C02 die (the "H"-class revision `RustyNES` has always
397 /// modeled). The OAMADDR (`$2003`) write-during-rendering OAM corruption is
398 /// **not** modeled, so the deterministic output is byte-identical to a build
399 /// without this feature.
400 #[default]
401 Rp2c02H,
402 /// Earlier RP2C02 die ("rev E+" in the nesdev notes). Additionally models the
403 /// OAMADDR (`$2003`) write-during-rendering OAM row-corruption glitch that
404 /// *Huge Insect* and a few other titles trip. Opt-in; changes emulated
405 /// behavior only for software that writes `$2003` mid-render.
406 Rp2c02G,
407}
408
409impl PpuRevision {
410 /// Whether this revision models the OAMADDR (`$2003`) write-during-rendering
411 /// OAM corruption glitch. Only [`Self::Rp2c02G`] does; the default returns
412 /// `false`, keeping the default build byte-identical.
413 #[must_use]
414 pub const fn models_oamaddr_corruption(self) -> bool {
415 matches!(self, Self::Rp2c02G)
416 }
417}
418
419/// v2.1.7 P5 — selectable power-up palette-RAM contents.
420///
421/// The 2C02's palette RAM is not cleared at power-on; different consoles (and
422/// thus different emulator authors' reference dumps) come up with different
423/// garbage. This is a documented power-up option, **default-off**: [`Default`]
424/// ([`Self::Zeroed`]) keeps `RustyNES`'s established all-zero power-up palette,
425/// so default rendering is byte-identical. It writes only `Ppu::palette_ram`,
426/// which is already part of the save-state snapshot, so it needs no
427/// snapshot-format change.
428#[derive(Clone, Copy, Debug, Eq, PartialEq, Hash, Default)]
429pub enum PaletteInit {
430 /// Default. All 32 palette-RAM bytes power up to `0x00` — `RustyNES`'s
431 /// established deterministic power-up state. Byte-identical.
432 #[default]
433 Zeroed,
434 /// The canonical "Blargg" power-up palette dump (the 32-byte pattern used by
435 /// blargg's NES and mirrored by `TriCNES`'s `BlarggPalette`). A documented,
436 /// deterministic known pattern for software that samples uninitialized
437 /// palette RAM before writing it. Opt-in.
438 Blargg,
439}
440
441/// The canonical "Blargg" power-up palette-RAM contents (32 bytes). Mirrors
442/// `TriCNES`'s `BlarggPalette` table (`Emulator.cs`) verbatim. Only applied when
443/// [`PaletteInit::Blargg`] is selected.
444const BLARGG_POWER_UP_PALETTE: [u8; 32] = [
445 0x09, 0x01, 0x00, 0x01, 0x00, 0x02, 0x02, 0x0D, 0x08, 0x10, 0x08, 0x24, 0x00, 0x00, 0x04, 0x2C,
446 0x09, 0x01, 0x34, 0x03, 0x00, 0x04, 0x00, 0x14, 0x08, 0x3A, 0x00, 0x02, 0x00, 0x20, 0x2C, 0x08,
447];
448
449/// 2C02 PPU.
450///
451/// `tick(bus)` advances one PPU dot. The PPU is the master clock;
452/// `rustynes-core` calls it three times per CPU cycle (NTSC).
453#[derive(Debug)]
454#[allow(clippy::struct_excessive_bools)] // PPU's many 1-bit latches are spec
455pub struct Ppu {
456 /// Region (governs frame structure).
457 pub(crate) region: PpuRegion,
458
459 // === CPU-facing register state ===
460 pub(crate) ctrl: PpuCtrl,
461 pub(crate) mask: PpuMask,
462 /// Two-stage delay pipeline of `mask` consumed exclusively by the
463 /// pre-render dot-339 odd-frame skip check. `mask_for_skip_check` is
464 /// the value seen *this* dot; `mask_skip_pipe1` is the staged value
465 /// that will become visible *next* dot. Both shift at the end of every
466 /// `advance_dot`. The total visible delay between a PPUMASK write and
467 /// the dot-skip detector is two PPU clocks — enough to compensate for
468 /// the lockstep bus model applying `cpu_write` at the *start* of a CPU
469 /// cycle (before the cycle's 3 PPU ticks) while real hardware latches
470 /// the write at φ2 (effectively the *end* of the cycle). Required by
471 /// blargg `ppu_vbl_nmi/10-even_odd_timing`; tests 1-9 of the same
472 /// corpus are unaffected because rendering is enabled long before
473 /// the boundary.
474 pub(crate) mask_for_skip_check: PpuMask,
475 pub(crate) mask_skip_pipe1: PpuMask,
476 /// True from the odd-frame skip until the dot it lands on has run:
477 /// scanline 0's dot 0, which the skip REPLACED rather than reached. The
478 /// `NESdev` PPU rendering page: the skip jumps "(339,261) to (0,0),
479 /// replacing the idle tick at the beginning of the first visible scanline
480 /// with the last tick of the last dummy nametable fetch". That tick
481 /// belongs to the dummy fetch, so it drives a nametable address, and the
482 /// dot-0 rule in `tick` (a visible line's dot 0 drives the background CHR
483 /// address) must not apply to it (T-MMC3-BG-A12).
484 ///
485 /// Snapshotted (`PPU_SNAPSHOT_VERSION` 12): a state saved between the skip
486 /// and that dot would otherwise restore into a dot 0 that raises A12 the
487 /// uninterrupted run did not.
488 pub(crate) dot0_replaced: bool,
489 pub(crate) status: PpuStatus,
490 /// `$2003` OAMADDR.
491 pub(crate) oam_addr: u8,
492 /// `$2007` PPUDATA read buffer.
493 pub(crate) data_buffer: u8,
494 /// `mc-ppu-2007-render-buffer`: the most recent VRAM data-bus value (every
495 /// rendering fetch updates it). During rendering a `$2007` read returns THIS
496 /// (the pipeline's current fetch byte), not a read at `v` — the model the
497 /// `AccuracyCoin` `$2007 Stress` test brackets (`$2007` read on every dot).
498 pub(crate) render_data_bus: u8,
499 /// `mc-ppu-2007-render-buffer`: PPU-dot countdown from a `$2007` read during
500 /// rendering to the `PPUDATA` state machine's `data_buffer` reload. `TriCNES`
501 /// reloads `PPU_ReadBuffer` via a latch cascade ~4 dots after read-END
502 /// (`Emulator.cs` `PPU_DATA_StateMachine`); with R1's fixed CPU<->PPU mod-4
503 /// phase the hardware landing dot is a constant +5 dots from the
504 /// register-read sample point (empirical W2 sweep winner: 170/170 stable
505 /// reads). Set at the read (`RENDER_BUFFER_DOT_DELAY`),
506 /// decremented once per dot in `Ppu::tick` (end of the fetch dispatch); at
507 /// 0 the buffer latches `render_data_bus` (the fetch cadence's latched bus
508 /// value — never a fresh VRAM read, so zero new A12/mapper events).
509 /// 0 = inactive. Serialized in the PPU snapshot v3 tail (W3-Stage-4,
510 /// 2026-06-10) so an in-flight reload survives a save-state restore.
511 pub(crate) ppudata_sm_countdown: u8,
512 /// `mc-ppu-2007-render-buffer`: the v-glitch increment of the in-flight
513 /// `$2007` read is still pending — performed at the `TStep` (= the countdown
514 /// landing dot, after the buffer reload), per `TriCNES`
515 /// `PPU_DATA_StateMachine_Half`. Serialized in the PPU snapshot v3 tail
516 /// (paired with `ppudata_sm_countdown`).
517 pub(crate) ppudata_v_inc_pending: bool,
518 /// `mc-ppu-2007-render-buffer`: raw (pre-h-flip) sprite pattern bytes
519 /// captured by `fetch_sprite_tile` per slot, so the per-dot sprite-fetch
520 /// read cadence (dots 257-320) feeds `render_data_bus` the PT-lo / PT-hi
521 /// values the PPU drove on the bus (the `$2007` buffer captures the raw
522 /// bus byte, not the flipped shifter contents). Serialized in the PPU
523 /// snapshot v3 tail.
524 pub(crate) spr_fetch_lo_raw: [u8; 8],
525 /// See [`Self::spr_fetch_lo_raw`].
526 pub(crate) spr_fetch_hi_raw: [u8; 8],
527 /// `mc-ppu-2007-render-buffer`: the nametable address latched by the ALE
528 /// of sprite slot 0's first garbage NT read (dot 257) — captured BEFORE
529 /// the dot-257 copy-hori, so the read at dot 258 uses the OLD `v` (the
530 /// only sprite-interval read that does). Serialized in the PPU snapshot
531 /// v3 tail (refreshed every rendered scanline).
532 pub(crate) ppudata_spr0_nt_addr: u16,
533
534 // === v2.0.3 (ADR 0030) octal-latch / 2-cycle-ALE multiplexed-bus model ===
535 // Now the ONLY PPU fetch path (promoted from the experimental flag in
536 // v2.0.3; the superseded v2.0.2 whole-dot stand-in was retired). In continuous
537 // play these fields self-heal within a scanline (they reload on the next fetch
538 // ALE), but a mid-render rollback/save-state checkpoint can capture them live —
539 // so they ARE serialized in the `PPU_SNAPSHOT_VERSION` v5 tail (added in v2.0.3)
540 // for netplay-rollback determinism. That bump is ADDITIVE (pre-v5 blobs upconvert
541 // to the inactive rest defaults), NOT an ADR-0028 save-state format-epoch break.
542 //
543 // Ported from TriCNES (`TriCNES/Emulator.cs`, MIT, commit 9199870),
544 // the AccuracyCoin author's own cycle-accurate C# emulator (a detailed
545 // sub-cycle CPU/PPU/APU/DMA state machine), which is the
546 // ground-truth oracle for the "ALE + Read" / "Hybrid Addresses" tests (the
547 // vendored Mesen2 build does NOT pass them — see the ADR 0030 campaign audit).
548 //
549 // The PPU multiplexes its low 8 VRAM address pins (PA0-7) with the 8 data
550 // pins (AD7-0). A 74LS373-class external octal latch (`octal_latch`) captures
551 // A7-A0 on the address-latch-enable (ALE) half of each 2-cycle VRAM access;
552 // on the read half the PPU drives only A13-A8 (the fetch address's high 6
553 // bits, `& 0x3F00`) and the latch supplies A7-A0. The *effective* read
554 // address is therefore the splice `(fetch_addr & 0x3F00) | octal_latch`
555 // (TriCNES `FetchPPU`:153). When those halves desync — a mid-fetch `$2006`
556 // high-byte update, or a `$2007`-read ALE that overlaps the fetch cadence and
557 // freezes the latch on a stale DATA byte — the PPU reads a "hybrid" address it
558 // never coherently drove.
559 /// 74LS373 low-address octal latch (A7-A0). Loaded on each fetch's ALE
560 /// with the driven address's low byte; frozen (goes stale) across a
561 /// `$2007`-read/background-fetch ALE overlap. Serialized in the PPU snapshot
562 /// v5 tail (v2.0.3): it self-heals within a scanline in continuous play, but a
563 /// mid-render rollback checkpoint needs it restored for determinism.
564 pub(crate) octal_latch: u8,
565 /// The multiplexed PPU address/data bus. On a fetch's ALE (even) dot the full
566 /// driven 14-bit address is written here; on the read (odd) dot the DATA byte
567 /// is written back into the low 8 bits (the AD7-0 pins), so the register always
568 /// reflects the true multiplexed bus value. The effective read address is the
569 /// splice `(address_bus & 0x3F00) | octal_latch`. Serialized in the PPU snapshot
570 /// v5 tail (v2.0.3) for mid-render rollback determinism (self-heals on the next
571 /// fetch ALE in continuous play).
572 pub(crate) address_bus: u16,
573 /// Set on a fetch's ALE (even) dot after the address is driven onto
574 /// [`Self::address_bus`]; consumed (cleared) by the matching read (odd) dot's
575 /// [`Self::ale_splice`]. Distinguishes a read that followed a real ALE (main
576 /// background dispatch, phases 0/2/4/6 → 1/3/5/7) from a read with NO preceding
577 /// ALE (the dot-337-340 garbage nametable fetches), so the latter stays
578 /// behavior-neutral (drives + latches coherently in-place rather than splicing
579 /// a stale bus).
580 pub(crate) ale_armed: bool,
581 /// One-shot: the `$2007`-read ALE overlap froze `octal_latch` on the read's
582 /// DATA byte, so the next background PATTERN fetch reads
583 /// `(PAR-high 6):(stale low 8)` (`$0F03` -> `$0FFF`) — the "ALE + Read"
584 /// corruption. Consumed by the next pattern (BG-lo/BG-hi) fetch: the latch is
585 /// frozen at the PPUDATA state-machine landing dot and carried to the next
586 /// pattern read via the natural ALE/read multiplex.
587 pub(crate) pattern_latch_stale: bool,
588 /// The delayed-`CopyV` countdown (`TriCNES`
589 /// `PPU_Update2006Delay`, `Emulator.cs:1684-1704`). A `$2006` second write
590 /// that lands DURING rendering does NOT copy `t -> v` immediately; it stages
591 /// this PPU-dot countdown instead. While it runs, the background fetch
592 /// cadence keeps advancing coarse-X (via `inc_hori_v` at phase 7) and the
593 /// per-group phase-0 nametable ALE keeps loading `octal_latch` with the
594 /// CURRENT (pre-copy) `v`'s NT-low byte — so by the time the countdown lands
595 /// the latch NATURALLY holds the one-tile-ahead low byte (`$19`), and the
596 /// landing's `address_bus = v` splice yields the hybrid address (`$2F19`)
597 /// with no `+1 coarse-X` reconstruction. 0 = inactive.
598 pub(crate) copy_v_delay: u8,
599
600 // === Internal scroll/address registers (loopy v/t/x/w) ===
601 /// 15-bit "current VRAM address".
602 pub(crate) v: u16,
603 /// 15-bit "temporary VRAM address" (latched scroll/PPUADDR target).
604 pub(crate) t: u16,
605 /// 3-bit fine X scroll.
606 pub(crate) x: u8,
607 /// 1-bit write toggle for `$2005` / `$2006`.
608 pub(crate) w: bool,
609
610 // === Memory ===
611 /// Console-side nametable VRAM (CIRAM, 2 KiB). Owned by the PPU; the
612 /// mapper exposes a per-cart `nametable_address` mirroring map via
613 /// [`PpuBus::nametable_address`] so the PPU can read/write CIRAM directly
614 /// without going through `bus.ppu_read/write` for `$2000-$3EFF`.
615 pub(crate) ciram: Box<[u8]>,
616 /// Object Attribute Memory: 64 sprites × 4 bytes.
617 pub(crate) oam: Box<[u8]>,
618 /// Secondary OAM: up to 8 sprites for the next scanline. Populated
619 /// during sprite evaluation in Sprint 2-3.
620 pub(crate) secondary_oam: [u8; 32],
621 /// Palette RAM: 32 entries, 6-bit each (high 2 bits open-bus on read).
622 pub(crate) palette_ram: [u8; 32],
623
624 // === v2.1.4 F2.3 optional OAM decay (opt-in, default-OFF) ===
625 /// Per-8-byte-row last-touch timestamp, in **CPU cycles** (`dot_counter / 3`).
626 /// OAM is 256 bytes = 32 rows of 8 bytes; `oam_decay_cycles[addr >> 3]` is the
627 /// CPU cycle at which row `addr >> 3` was last refreshed by an OAM read/write.
628 /// Only consulted when [`Self::oam_decay_enabled`] is set AND the region is
629 /// NTSC/Dendy; otherwise it is dead state that never influences a read. Mirrors
630 /// Mesen2's `_oamDecayCycles` (`Core/NES/NesPpu.cpp`).
631 ///
632 /// Serialized as a *relative age* in the PPU snapshot v7 tail (see
633 /// `snapshot.rs`): the absolute timestamps reference the free-running,
634 /// **un-serialized** `dot_counter`, so a raw absolute value would be meaningless
635 /// after a rollback/restore rebased that counter. Storing `now - timestamp`
636 /// (and reconstructing `now - age` on load, relative to the live counter) keeps
637 /// a run-ahead / netplay `snapshot`→`restore` byte-identical to the forward run.
638 ///
639 /// Field POSITION here is not performance-relevant, and this was measured
640 /// rather than assumed (v2.3.1 G2, `docs/performance.md`): neither adding
641 /// `#[repr(C)]` nor moving this 256-byte cold array to the end of the struct
642 /// produced a reproducible change on any workload. `Ppu` is ~2.8 KB and stays
643 /// L1-resident across a frame, so layout has little left to buy.
644 pub(crate) oam_decay_cycles: [u64; 32],
645 /// Master enable for the OAM-decay model. **`false` by default** — a frontend /
646 /// config knob (re-applied on load like `region` / `active_palette`), NOT part
647 /// of the save-state. While `false`, every decay hook early-returns, so OAM
648 /// reads/writes touch neither this flag's siblings nor `oam_decay_cycles`, and
649 /// the framebuffer/audio/replay output is byte-identical to a decay-free build.
650 pub(crate) oam_decay_enabled: bool,
651
652 // === Open-bus latch (for $2000-$3FFF) ===
653 /// Most recent value driven onto the PPU bus by any register access.
654 pub(crate) open_bus: u8,
655 /// Per-bit-group decay counters (in CPU cycles) until each bit group of
656 /// the open-bus latch reads as 0. Three groups, each with its own timer:
657 /// `[0]` bits 0-4, `[1]` bit 5, `[2]` bits 6-7.
658 /// Required by `ppu_open_bus.nes` tests 7 and 9, which assert that some
659 /// reads refresh only a subset of the bit groups (e.g., reading $2002
660 /// must not refresh the low 5 bits' decay timer; palette $2007 reads
661 /// must not refresh the high 2 bits' decay timer).
662 pub(crate) open_bus_decay: [u32; 3],
663
664 // === NMI line + frame counter ===
665 /// `true` while the PPU is asserting NMI.
666 pub(crate) nmi_line: bool,
667 /// True for one frame after a `cpu_read_register($2002)` race so we
668 /// suppress the VBL flag set + NMI for that frame (per
669 /// `ppu_vbl_nmi/06-suppression.nes`). Toggled on the cycle the read
670 /// hits at scanline 241 dot 0 / dot 1.
671 pub(crate) suppress_vbl_this_frame: bool,
672 /// Last-observed A12 level, for edge-triggered notifications.
673 pub(crate) last_a12_level: bool,
674
675 // === Scanline FSM ===
676 /// Current dot (0..=340).
677 pub(crate) dot: u16,
678 /// Current scanline (-1 in pre-render, 0..=239 visible, 240 post-render,
679 /// 241..=260/310 vblank). Stored as i16 to allow temporary -1.
680 pub(crate) scanline: i16,
681 /// Frame counter (for odd-frame skip).
682 pub(crate) frame: u64,
683 /// `frame_complete` latch — set to `true` on the dot the PPU finishes
684 /// a frame; consumed by the run loop and cleared on next read.
685 pub(crate) frame_complete: bool,
686
687 // === Power-on / reset masking window ===
688 /// CPU cycles remaining in the post-reset masking window. While > 0,
689 /// writes to PPUCTRL/PPUMASK/PPUSCROLL/PPUADDR are silently ignored
690 /// (reads still work).
691 pub(crate) post_reset_mask_remaining: u32,
692
693 // === Background fetch + shift register state ===
694 /// Latched nametable byte from the current 8-cycle fetch group.
695 pub(crate) nt_latch: u8,
696 /// Latched attribute byte (palette) from the current 8-cycle fetch group.
697 pub(crate) at_latch: u8,
698 /// Latched BG pattern low byte from the current 8-cycle fetch group.
699 pub(crate) bg_lo_latch: u8,
700 /// Latched BG pattern high byte from the current 8-cycle fetch group.
701 pub(crate) bg_hi_latch: u8,
702 /// 16-bit BG pattern low shift register.
703 pub(crate) bg_shift_lo: u16,
704 /// 16-bit BG pattern high shift register.
705 pub(crate) bg_shift_hi: u16,
706 /// 16-bit attribute low shift register.
707 ///
708 /// Mirrors the 16-bit BG pattern shifters exactly: at each 8-dot
709 /// reload the latched attribute bit is expanded to a full byte
710 /// (`0x00` or `0xFF`) into bits 0-7, shifted left by 1 after each
711 /// emit, and shifted left by 8 at the pre-fetch boundary (dots 328 /
712 /// 336). Keeping it 16-bit (not the prior 8-bit + 1-bit-feed model)
713 /// is what keeps the attribute in lockstep with the pattern bits
714 /// through the dots 321-336 pre-fetch region, where `shift_bg` does
715 /// not run and only the explicit `<<= 8` advances the registers.
716 pub(crate) at_shift_lo: u16,
717 /// 16-bit attribute high shift register. See [`Self::at_shift_lo`].
718 pub(crate) at_shift_hi: u16,
719 /// Optional per-tile extended attribute (MMC5 `ExGrafix`). Latched at
720 /// the NT-byte fetch boundary; consumed by AT / BG-low / BG-high
721 /// fetches in the same 8-dot group.
722 pub(crate) ex_attr_latch: Option<ExAttribute>,
723 /// Optional vertical split-screen state (MMC5 `$5200`-`$5202`). Latched
724 /// at the NT-byte fetch boundary; consumed by AT / BG-low / BG-high
725 /// fetches in the same 8-dot group. When `Some`, the BG fetches use the
726 /// alt region's nametable address, attribute address, fine-Y, and CHR
727 /// bank instead of the values derived from `v`.
728 pub(crate) bg_split_latch: Option<BgSplitState>,
729
730 // === Sprite rendering state ===
731 /// Per-sprite shift registers (low + high pattern).
732 pub(crate) spr_shift_lo: [u8; 8],
733 pub(crate) spr_shift_hi: [u8; 8],
734 /// Per-sprite latched attribute byte.
735 pub(crate) spr_attr: [u8; 8],
736 /// Per-sprite X-coordinate counter.
737 pub(crate) spr_x: [u8; 8],
738 /// v2.0 (ppu-sprite-shifter-counter): per-sprite persistent "halted" latch.
739 /// Set when the X-counter reaches 0 (the sprite is drawing) and PERSISTS
740 /// across a render-disable and the frame boundary; re-armed to "counting"
741 /// at dot 339 for the loaded slots. Carries the Stale Sprite Shift Regs
742 /// t5/6 behavior (a reloaded-but-halted sprite draws on the next rendering
743 /// re-enable). Default build: absent (legacy `spr_x == 0` predicate).
744 pub(crate) spr_halted: [bool; 8],
745 /// Number of sprites loaded for the current scanline.
746 pub(crate) spr_count: u8,
747 /// `true` if sprite 0 is in the current scanline's sprite line-up.
748 pub(crate) spr_zero_in_line: bool,
749
750 // === Per-dot sprite-evaluation FSM state ===
751 /// Sprite-eval read latch: byte read from primary OAM on odd cycles
752 /// (1, 3, 5, ...) of dots 65-256, consumed by the immediately-following
753 /// even-cycle write into secondary OAM.
754 pub(crate) sprite_eval_read_latch: u8,
755 /// Primary-OAM sprite index 0..=63 walked during dots 65-256.
756 pub(crate) sprite_eval_n: u8,
757 /// Per-sprite byte index 0..=3 walked during dots 65-256 (drives the
758 /// buggy `n+m` increment when overflow detection mode is active).
759 pub(crate) sprite_eval_m: u8,
760 /// Number of in-range sprites found so far in this scanline's eval pass.
761 pub(crate) sprite_eval_found: u8,
762 /// Write index into `secondary_oam` (0..=31). Tracks how many bytes the
763 /// per-dot FSM has committed so far.
764 pub(crate) sprite_eval_sec_idx: u8,
765 /// `true` when the current sprite (the one whose `y` byte just tested
766 /// in-range) is still being copied — bytes 1, 2, 3 land in subsequent
767 /// even-dot writes.
768 pub(crate) sprite_eval_copying: bool,
769 /// `true` when eval has exhausted primary OAM (n wrapped past 63) or
770 /// overflow has been detected — remaining dots 65-256 idle out.
771 pub(crate) sprite_eval_done: bool,
772 /// `true` when 8 in-range sprites have been latched and the FSM is
773 /// in overflow-detection mode (buggy `n+m` increment active).
774 pub(crate) sprite_eval_overflow_search: bool,
775 /// Eval-side latch for "sprite 0 is in the line being evaluated."
776 /// Set during the current scanline's eval pass (dots 65..=256) when
777 /// sprite 0 lands in-range; committed to [`Self::spr_zero_in_line`]
778 /// at dot 256 alongside [`Self::spr_count`]. Keeping the eval-side
779 /// latch separate from the rendering-side flag ensures the FSM
780 /// doesn't trample the CURRENT scanline's sprite-0-hit signal while
781 /// it's still being read by the dots 1..=256 sprite-pixel evaluator.
782 pub(crate) sprite_eval_zero_found: bool,
783 /// Phase 3a flag — tracks whether current scanline's eval is on
784 /// its FIRST iteration (PPU cycle 66, first y-test). Set at
785 /// dot 0 of each visible scanline; cleared after the first y-test
786 /// fires (in-range or not). Per Mesen2 `ProcessSpriteEvaluation`
787 /// line 1040-1044, sprite-zero fires IFF the FIRST y-test is in
788 /// range — not "first in-range sprite found". When OAMADDR is 0
789 /// at eval start and OAM[0].y is in range, this matches the legacy
790 /// `n == 0` check. When OAMADDR != 0, this fires on whichever
791 /// sprite the start position points to (sprite at OAMADDR / 4)
792 /// if its y is in range, else NO sprite-zero is detected.
793 pub(crate) sprite_eval_first_iter: bool,
794
795 /// v2.0 Tier 1.2 — isolated OAM-data-bus model of the `NESdev`-documented PPU
796 /// sprite-evaluation datapath (`NESdev` wiki "PPU sprite evaluation"). These
797 /// fields exist ONLY under `ppu-oam-data-bus` and are read solely by `$2004`
798 /// during rendering — the rendering / sprite-zero / overflow / MMC3
799 /// sprite-fetch FSM uses `secondary_oam` + `sprite_eval_*` + `spr_*`, all
800 /// untouched. `oam_bus_copybuffer` is the value `$2004` returns while the
801 /// screen is drawn (the byte currently on the OAM data bus).
802 ///
803 /// Provenance: the OAM-data-bus and sprite-evaluation model is **derived
804 /// from Mesen2's `NesPpu.cpp`** (`ProcessSpriteEvaluation` / `ReadSpriteRam`),
805 /// GPL-3.0-or-later. See NOTICE and docs/originality-and-provenance.md (Section 1).
806 pub(crate) oam_bus_copybuffer: u8,
807 /// Parallel secondary OAM (the 32-byte sprite line buffer) for the bus model only.
808 pub(crate) oam_bus_secondary: [u8; 32],
809 /// Eval-pointer sprite index (0..=63) — which of the 64 primary sprites is examined.
810 pub(crate) oam_bus_addr_h: u8,
811 /// Eval-pointer byte-in-sprite (0..=3) — Y / tile / attr / X.
812 pub(crate) oam_bus_addr_l: u8,
813 /// Write index into the parallel secondary OAM.
814 pub(crate) oam_bus_secondary_addr: u8,
815 /// Primary OAM fully scanned / wrapped for this scanline.
816 pub(crate) oam_bus_copy_done: bool,
817 /// Currently copying an in-range sprite.
818 pub(crate) oam_bus_sprite_in_range: bool,
819 /// The 8-sprite-overflow PPU-bug countdown.
820 pub(crate) oam_bus_overflow_counter: u8,
821
822 /// OAM-corruption model — faithful port of `TriCNES`'s eval-pointer
823 /// machinery (`Emulator.cs` `PPU_Render_SpriteEvaluation` lines
824 /// 2664-2770 + `CorruptOAM` lines 2635-2651). Replaces the earlier
825 /// Mesen2 `_corruptOamRow` row-flag model (`dot >> 1` index), which
826 /// Mesen ships OFF by default (`EnablePpuOamRowCorruption=false`) and
827 /// documents as unfinished. The bug it fixes: SMB3 (MMC3) toggles
828 /// PPUMASK mid-visible-scanline to split its HUD; NMI/DMA jitter
829 /// shifts the disable dot, and the raw-dot row index intermittently
830 /// landed on Mario's OAM row (offset 40), wiping his sprite.
831 ///
832 /// `TriCNES` model: when rendering is disabled (1 -> 0) DURING sprite
833 /// evaluation (dots 1-64, secondary-OAM clear, NOT the pre-render
834 /// line), the corruption is DEFERRED — `oam_corruption_pending` is
835 /// set and `oam_corruption_index` captures the live secondary-OAM
836 /// write pointer (`OAM2Address`) at that instant. When rendering
837 /// RE-ENABLES (or at the pre-render line), one OAM "row" of 8 bytes
838 /// is replaced from row 0: `oam[index*8 + i] = oam[i]` for i in
839 /// 0..8 (index 0x20 wraps to 0), and `secondary_oam[index] =
840 /// secondary_oam[0]`.
841 ///
842 /// `oam2_addr` is the `OAM2Address` analogue maintained across the
843 /// dots 1-64 clear window (our dots 65-256 active eval already walks
844 /// `sprite_eval_sec_idx`, but the SMB3 HUD-split disable lands in the
845 /// clear window where `sprite_eval_sec_idx` is held at 0, so the
846 /// dedicated pointer is required to capture the right index).
847 ///
848 /// `oam_corruption_disabled` / `_instant` mirror `TriCNES`'s
849 /// `PPU_OAMCorruptionRenderingDisabledOutOfVBlank` (1-dot-delayed,
850 /// armed by the `$2001` write-delay) and `..._Instant` (the
851 /// data-bus-immediate path: OAM eval observes the disable the same
852 /// cycle). The disable edge is captured into `pending`/`index`
853 /// during the dots 1-64 eval window; the actual corruption is
854 /// committed at re-enable / pre-render.
855 ///
856 /// None of these fields are persisted in the PPU snapshot — like the
857 /// rest of the per-dot sprite-eval FSM state (`sprite_eval_*`), they
858 /// re-derive within a scanline/frame, matching the prior row-flag
859 /// OAM-corruption model (also un-snapshotted).
860 pub(crate) oam_corruption_pending: bool,
861 pub(crate) oam_corruption_index: u8,
862 pub(crate) oam_corruption_disabled: bool,
863 pub(crate) oam_corruption_disabled_instant: bool,
864 /// `OAM2Address` analogue: the secondary-OAM write pointer as it
865 /// walks the dots 1-64 secondary-OAM clear window. Reset to 0 at the
866 /// dot-1 boundary and incremented once per even clear dot, masked to
867 /// 0x1F, exactly as `TriCNES` drives `OAM2Address` during dots 1-64.
868 pub(crate) oam2_addr: u8,
869 /// `OAM2Address` as a LIVE counter across sprite fetch (dots 257-320),
870 /// and the "OAM2 Overflowed" flag that freezes it.
871 ///
872 /// Both rules are stated outright by `AccuracyCoin`'s own source, which is
873 /// the specification here (MIT; stimulus, not a reference implementation):
874 ///
875 /// > When OAM2 is full, the PPU prevents further increments of the OAM2
876 /// > Address, so the OAM2 Address is frozen at index 0. This flag ... is
877 /// > cleared if rendering is enabled during dots 63, 255, and 339.
878 ///
879 /// The counter advances on EVEN dots (32 bytes across the 64-dot fetch).
880 /// That yields 33 candidate increments over dots 256..=320, and the flag
881 /// is what reconciles it: the 32nd wraps 0x1F -> 0 AND raises the flag,
882 /// which suppresses the 33rd, leaving the address resting at 0. So the
883 /// documented "`$2004` during dots 321-340 reads OAM2[0]" is not a special
884 /// case — it is the ordinary end state of the counter, and the flag is
885 /// load-bearing for it. Both `AccuracyCoin` `Misaligned OAM2 Address` and
886 /// `Frozen OAM2 Increment` are explained by this one mechanism.
887 ///
888 /// **Snapshotted, in the `PPU_SNAPSHOT_VERSION` 9 tail** — unlike
889 /// `oam2_addr` and the rest of the per-dot sprite-eval FSM, which re-derive
890 /// within a scanline. This one cannot: the whole point of the live counter
891 /// is that it REMEMBERS increments it did not take while rendering was off,
892 /// so a restore that re-derived it would discard exactly the state the
893 /// counter exists to carry. A pre-v9 blob has no such field and restores to
894 /// `0` / `false` / `false`, which is the power-on state.
895 pub(crate) oam2_fetch_addr: u8,
896 pub(crate) oam2_overflowed: bool,
897 /// The freeze flag LATCHED at the start of sprite fetch (dot 257).
898 ///
899 /// Read instead of `oam2_overflowed` by the sprite loader, because in the
900 /// ordinary case the counter wraps and raises the flag near the END of the
901 /// fetch window (dot 318) -- reading the live flag would then retroactively
902 /// force the last slot to OAM2[0] and corrupt every game's final sprite.
903 /// The hardware behaviour under test is about the flag's state going IN.
904 pub(crate) oam2_fetch_frozen: bool,
905 /// Previous-tick rendering-enabled state — tracks the rising /
906 /// falling edge of `mask.rendering_enabled()` so the 1->0 edge
907 /// BG-shifter fix-up fires on the correct transition.
908 pub(crate) prev_rendering_enabled: bool,
909 /// v2.0 (ported from branch `ae30785`) — 1-PPU-dot-delayed rendering-enabled
910 /// gate. Per Mesen2 `NesPpu::UpdateState`: a `$2001` write toggling
911 /// `SHOW_BG|SHOW_SPRITE` takes effect on the rendering pipeline one PPU dot
912 /// later (the `mask` bit-fields update immediately for pixel output; this
913 /// delayed copy gates the fetch/shift/sprite-eval pipeline). When rendering
914 /// is stable this equals `mask.rendering_enabled()`, so only mid-scanline
915 /// `$2001` toggles observe the delay. The `ppu-sprite-shifter-counter`
916 /// feature reads it (via `rendering_gate`); the default build uses the
917 /// immediate value, so flag-off is byte-identical. Updated at tick end.
918 pub(crate) rendering_enabled_delayed: bool,
919
920 /// Two-dots-ago rendering value, for the v2.6.18 depth knob only.
921 #[cfg(feature = "phi2-write-sweep")]
922 pub(crate) render_gate_prev2: bool,
923
924 /// Rendering as it stood TWO dots ago -- the gate the dot-256 vertical
925 /// increment reads, and the only consumer that needs a second stage.
926 ///
927 /// v2.6.18. `AccuracyCoin`'s `Frozen OAM2 Increment` test 2 states the rule
928 /// this models: *"Rendering is enabled on dot 256, but the PPU's vertical
929 /// scroll is NOT incremented."* That sentence only parses if the enable IS
930 /// on dot 256 -- so an enable in force from the start of dot 256 must not
931 /// fire dot 256's increment, and the increment therefore sees the mask one
932 /// dot deeper than [`Ppu::rendering_enabled_delayed`] does.
933 ///
934 /// Measured rather than fitted: a `$2001` write whose effect lands during
935 /// dot N is in force from the start of N+1, and the ROM over-determines the
936 /// depth (the dot-256 increment must not fire, the dot-257 latch must), so
937 /// `d = 257 - 255 = 2`. Both neighbouring depths fail -- one dot shallower
938 /// is byte-identical to the unfixed build, one dot deeper breaks test 2.
939 pub(crate) rendering_enabled_delayed2: bool,
940
941 /// The dot-339 sprite re-arm, deferred past scanline 0's first pixel
942 /// because the odd-frame skip removed the dot at which hardware sees it.
943 ///
944 /// v2.9.5. The shifters are told to start counting on dot 339 and see the
945 /// signal a dot late, at 340. When an odd frame skips pre-render dot 340,
946 /// every loaded shifter therefore starts scanline 0 still in the drawing
947 /// state. It outputs its first pixel at X=0 and shifts, then counts one dot
948 /// late, so pixels 1-7 land where they always do and only the first moves
949 /// (`forums.nesdev.org/viewtopic.php?t=26291`, from `Visual2C02` analysis).
950 /// This is what `AccuracyCoin`'s `Sprites On Scanline 0` reads as a
951 /// composite 2C02: code 1, where the missing alternation had read as an
952 /// RGB PPU, code 2.
953 ///
954 /// Set by the skip in `advance_dot` and cleared after pixel 0 in
955 /// `emit_pixel`. It lives across the frame boundary, where run-ahead and
956 /// save states snapshot, so it is serialized (`PPU_SNAPSHOT_VERSION` 11).
957 pub(crate) spr_rearm_deferred: bool,
958
959 /// v2.6.18 (`phi2-write-sweep` only): a shared four-stage `$2001` history,
960 /// newest first, that each swept consumer gate reads at its OWN depth.
961 ///
962 /// The first version of `OAM2_GATE_LAG` borrowed the odd-frame-skip
963 /// pipeline, which capped the reachable depth at two AND coupled two
964 /// unrelated consumers. The measurement that mattered lives past that cap:
965 /// at the shipped placement `Frozen OAM2 Increment` closes only with the
966 /// OAM2 gate deeper than the skip pipeline can express.
967 #[cfg(feature = "phi2-write-sweep")]
968 pub(crate) sweep_mask_history: [PpuMask; 4],
969
970 /// v2.0 Phase 6 (`mc-ppu-subpos`): the analog `$2001` BG-shift-register
971 /// RELOAD delay. The shifter reload gates on `bg_reload_render`, which tracks
972 /// the live `self.mask` rendering-enable bit EXCEPT during the
973 /// `MASK_WRITE_DELAY`-dot window after a `$2001` write, where it stays frozen
974 /// at its prior value (`TriCNES` gates the fetch/reload on
975 /// `PPU_Mask_Show*_Delayed` while the per-half-dot SHIFT runs on the
976 /// IMMEDIATE mask). So on a render re-enable the shifter advances for
977 /// `MASK_WRITE_DELAY` dots (injecting the serial-in '1') BEFORE the reload
978 /// resumes -> one reload is SKIPPED and the accumulated '1's reach the output
979 /// (BG Serial In), without perturbing the sprite/pixel/shift path. Because it
980 /// re-syncs to the live mask whenever settled, a direct mask set (unit tests,
981 /// save-state restore) leaves it consistent. `mask_write_delay` is the
982 /// remaining freeze countdown (0 = settled). Serialized in the PPU
983 /// snapshot v3 tail (W3-Stage-4, 2026-06-10) so an in-flight freeze
984 /// survives a save-state restore.
985 pub(crate) bg_reload_render: bool,
986 pub(crate) mask_write_delay: u8,
987
988 /// v1.4.0 Workstream F (F1) — scanline-stable rendering-classification
989 /// cache. `visible` / `pre_render` / `render_line` are pure functions of
990 /// `self.scanline` + `self.region`, so they only change when the scanline
991 /// advances. The hot per-dot `tick` recomputes them ~7 branches deep
992 /// 89,342 times/frame; instead we recompute them once when the scanline
993 /// changes (detected via the `flags_cached_scanline` sentinel) and read the
994 /// cached copies on every other dot. Byte-identical by construction (same
995 /// values, computed less often) and self-healing across reset / save-state
996 /// restore (the sentinel starts mismatched, forcing a recompute on the
997 /// first tick). NOT part of the PPU snapshot — pure derived data.
998 pub(crate) cached_visible: bool,
999 pub(crate) cached_pre_render: bool,
1000 pub(crate) cached_render_line: bool,
1001 /// v2.2.3 P2 — `true` on an **idle** line: not visible, not pre-render, and
1002 /// not the VBL-set line (`vblank_start_line`). On NTSC that is line 240
1003 /// (post-render) plus lines 242..=260 — 20 of 262. Such a line issues no
1004 /// fetch, emits no pixel, runs no sprite evaluation, and raises no event, so
1005 /// its dots are eligible for [`Self::tick_idle_line_fast`]. Derived from
1006 /// `scanline` + `region` exactly like the three flags above, and keyed by
1007 /// the same [`Self::flags_cached_scanline`] sentinel.
1008 #[cfg(feature = "ppu-idle-line-fast")]
1009 pub(crate) cached_idle_line: bool,
1010 /// Sentinel: the scanline `cached_*` were last computed for. `i16::MIN`
1011 /// (an impossible scanline) forces a recompute on the first tick.
1012 pub(crate) flags_cached_scanline: i16,
1013
1014 /// Active output palette. Defaults to the 2C02 composite palette so normal
1015 /// NES/Famicom rendering is byte-for-byte unchanged; set to one of the RGB
1016 /// variants for Vs. System / PlayChoice-10 carts (see [`Ppu::set_palette`]).
1017 /// Construction-time configuration only — never mutated during emulation, so
1018 /// it is intentionally NOT part of the PPU save-state snapshot (it is
1019 /// re-derived from the cartridge header on load).
1020 pub(crate) active_palette: crate::palette::PpuPalette,
1021 /// v2.8.0 Phase 4 — precomputed `(emphasis bits << 6) | color` → RGBA8
1022 /// lookup (8 emphasis combinations × 64 colors = 512 entries, 2 KiB).
1023 /// Built from the same pure [`crate::palette::palette_color_to_rgba`]
1024 /// the per-pixel path used to call, so it is byte-identical by
1025 /// construction; rebuilt whenever [`Ppu::set_palette`] changes the
1026 /// active palette. Saves a palette-variant match + emphasis branches
1027 /// per emitted pixel (61,440/frame). NOT part of the save-state
1028 /// (derived data).
1029 pub(crate) rgba_lut: [[u8; 4]; 512],
1030 /// v1.1.0 beta.1 (T-110-A3) — optional custom 64-entry base palette from a
1031 /// loaded `.pal` file. `None` (default) = use the built-in palette for the
1032 /// active [`crate::palette::PpuPalette`], so default rendering is
1033 /// byte-identical. When `Some`, [`Self::rgba_lut`] is built from it via the
1034 /// composite emphasis model. A frontend presentation override — NOT part of
1035 /// the save-state (it persists across save/load like the active palette).
1036 pub(crate) custom_palette: Option<[[u8; 3]; 64]>,
1037 /// True when this is a 2C05 PPU: `$2000`/`$2001` are swapped and `$2002`
1038 /// returns the 2C05 sub-variant identifier in its low bits. Default false
1039 /// (a 2C02 / 2C03 / 2C04, none of which swap or report an id).
1040 pub(crate) is_2c05: bool,
1041 /// 2C05 sub-variant `$2002` identifier byte (e.g. `$3D` for 2C05-02). Only
1042 /// consulted when [`Self::is_2c05`] is true. Combined into the low 5 bits
1043 /// of a `$2002` read per nesdev "PPU registers" §2C05 identifier.
1044 pub(crate) id_2c05: u8,
1045
1046 /// v2.1.7 P5 — selected 2C02 die revision. Gates the OAMADDR (`$2003`)
1047 /// write-during-rendering OAM corruption glitch (see [`PpuRevision`]). The
1048 /// [`PpuRevision::default`] ([`PpuRevision::Rp2c02H`]) models NO extra
1049 /// corruption, so the default build is byte-identical. Construction / config
1050 /// only — never mutated by emulation and, like [`Self::active_palette`] /
1051 /// [`Self::region`], re-applied on load rather than serialized in the
1052 /// snapshot (the corruption *state* it can arm — `oam_corruption_pending` /
1053 /// `oam_corruption_index` — IS in the v6 snapshot tail, so an armed
1054 /// corruption still round-trips).
1055 pub(crate) die_revision: PpuRevision,
1056 /// v2.1.7 P5 — the power-up palette-RAM contents selected for this PPU (see
1057 /// [`PaletteInit`]). Stored so a power-cycle can re-apply it after the PPU is
1058 /// reconstructed. The [`PaletteInit::default`] ([`PaletteInit::Zeroed`])
1059 /// leaves palette RAM all-zero (the established default), keeping default
1060 /// rendering byte-identical. Config, not serialized (it writes
1061 /// [`Self::palette_ram`], which the snapshot already carries).
1062 pub(crate) power_up_palette: PaletteInit,
1063
1064 /// Framebuffer (RGBA8). Filled by Sprint 2-2/2-3 rendering.
1065 pub(crate) framebuffer: Box<[u8]>,
1066
1067 /// v1.1.0 beta.1 (T-110-A1) — parallel per-pixel **palette-index**
1068 /// framebuffer for the true composite `NES_NTSC` filter. Each entry is the
1069 /// 9-bit `(emphasis << 6) | colour_index` value (0..=511) written in the
1070 /// same emit path as [`Self::framebuffer`], so it is a faithful index-space
1071 /// mirror of the RGBA output. The frontend uploads it as an `R16Uint`
1072 /// texture and reconstructs the composite signal in a shader. Purely an
1073 /// output buffer: it changes no logical state, so the determinism /
1074 /// `AccuracyCoin` contract is unaffected. Unlike `framebuffer` (which IS in
1075 /// the save-state), this and `dot_counter` / `frame_ntsc_phase` are NOT
1076 /// serialized — they are regenerated on the next emitted frame, so a state
1077 /// loaded while paused shows correct NTSC from the first frame after resume.
1078 pub(crate) index_framebuffer: Box<[u16]>,
1079 /// How many dots took the specialized fast path (`ppu-fetch-trace` only).
1080 ///
1081 /// Not telemetry. It exists so a test can assert it EXERCISED the fast path
1082 /// rather than passing because the path was never entered — the failure this
1083 /// project keeps finding, where a check agrees about something it never
1084 /// reached. A review of #450 claimed the fast path bypasses the fetch trace;
1085 /// the test refuting that is worthless unless it can show the path ran, and
1086 /// this is how it shows it.
1087 #[cfg(feature = "ppu-fetch-trace")]
1088 pub fast_path_hits: u64,
1089
1090 /// Free-running PPU master-cycle counter (one increment per [`Self::tick`]),
1091 /// the basis for the per-frame NTSC colour phase. Output-only / cosmetic
1092 /// (drives only the optional NTSC filter's dot-crawl); not part of the
1093 /// save-state. Wraps harmlessly.
1094 pub(crate) dot_counter: u64,
1095
1096 /// NTSC composite colour phase snapshotted at each frame boundary, the
1097 /// per-frame `videoPhase` consumed by the `NES_NTSC` filter (the shader
1098 /// derives the per-scanline / per-pixel phase from this base). `0..=2` on
1099 /// NTSC; on PAL/Dendy (no 3-phase crawl) it is the frame parity (`0..=1`).
1100 /// Cosmetic; not part of the save-state.
1101 pub(crate) frame_ntsc_phase: u8,
1102
1103 /// v1.7.0 "Forge" Workstream F3 — PPU extra-scanlines overclock.
1104 ///
1105 /// Number of EXTRA blank scanlines to insert into the vblank period each
1106 /// frame (immediately before the pre-render line), at the existing dot
1107 /// resolution (Mesen2 `UpdateTimings`). These lines render nothing, emit no
1108 /// pixels, set/clear no PPU flags, and fire no VBL/NMI/A12 events — they are
1109 /// pure additional CPU run-time per frame, giving games more compute headroom
1110 /// without altering the visible image. **Off by default (`0`)**; the
1111 /// `advance_dot` insertion path is entirely guarded by `extra_scanlines != 0`,
1112 /// so at the default this field changes nothing and the frame is
1113 /// byte-identical to stock. Distinct from the CPU-multiplier overclock
1114 /// (`Nes::set_cpu_overclock`, v3.1.0, in the bus). A frontend config knob,
1115 /// NOT part of the save-state (re-applied by the frontend on restore, like
1116 /// `region` / `active_palette`).
1117 pub(crate) extra_scanlines: u16,
1118 /// v3.1.0 (`T-SPRITE-LIMIT`, FE-02): draw the sprites beyond the eighth on
1119 /// a scanline. **Render-only**: sprite evaluation, secondary OAM, the
1120 /// overflow flag, sprite-0 hit and every real sprite fetch (with its A12
1121 /// edges) are untouched. The extra sprites' patterns are read after the
1122 /// eight real fetches, through [`PpuBus::chr_reads_are_pure`] boards only,
1123 /// with no A12 notification, and they draw behind all eight hardware
1124 /// sprites (a higher OAM index is a lower priority). Off by default;
1125 /// configuration, carried across a power cycle by
1126 /// [`Self::adopt_settings_from`] and in movies / netplay by the core's
1127 /// `HardwareOptions`.
1128 pub(crate) sprite_limit_disabled: bool,
1129 /// v3.1.0 — the extra sprites fetched for the next scanline (snapshot v13):
1130 /// how many, and for each the h-flip-applied pattern bytes, attributes and
1131 /// X. Always 0 while `sprite_limit_disabled` is off.
1132 pub(crate) spr_extra_count: u8,
1133 pub(crate) spr_extra_lo: [u8; MAX_EXTRA_SPRITES],
1134 pub(crate) spr_extra_hi: [u8; MAX_EXTRA_SPRITES],
1135 pub(crate) spr_extra_attr: [u8; MAX_EXTRA_SPRITES],
1136 pub(crate) spr_extra_x: [u8; MAX_EXTRA_SPRITES],
1137 /// v1.7.0 F3 — countdown of extra blank scanlines remaining for the CURRENT
1138 /// frame's vblank insertion. Loaded from [`Self::extra_scanlines`] when the
1139 /// PPU reaches the insertion point and decremented one extra line at a time.
1140 /// `0` when no insertion is in flight. Snapshotted (snapshot v4) so a
1141 /// save-state taken mid-insertion restores the in-flight countdown rather
1142 /// than resuming as `0` and desyncing. The configured count itself
1143 /// (`extra_scanlines`) stays a non-persisted frontend knob, re-applied on
1144 /// restore. At the default `extra_scanlines == 0` this is always `0`.
1145 pub(crate) extra_lines_remaining: u16,
1146
1147 /// v2.1.8 A1 — enable the specialized straight-line per-dot fast path for
1148 /// the common visible-scanline BG-render window (see [`Self::tick`] and
1149 /// `docs/performance.md`). When `true`, visible scanline dots `1..=256`
1150 /// whose per-dot state is provably "undisturbed" (no pending `$2006`
1151 /// copy-V, no PPUMASK write-delay, no PPUDATA state machine in flight, no
1152 /// armed/pending OAM-corruption, warm scanline classification cache,
1153 /// stable rendering-enable) are dispatched to
1154 /// [`Self::tick_visible_render_fast`], which executes the identical
1155 /// helper sequence with the statically-dead event/bookkeeping branches
1156 /// pruned. Any disturbance drops instantly back to the exact path.
1157 /// **Not serialized** — a frontend/config knob re-applied on restore.
1158 ///
1159 /// **Default `true` since the v2.2.3 performance pass (was default-OFF for
1160 /// v2.1.8 .. v2.2.2).** A1 shipped this off deliberately: it was that
1161 /// roadmap's highest-risk item, and keeping it off left the shipped build
1162 /// byte-identical while the differential test and the oracle suites proved
1163 /// correctness. Both conditions A1 named for promotion are now met —
1164 ///
1165 /// * **byte-identity**, held continuously since v2.1.8 by
1166 /// `crates/rustynes-test-harness/tests/fast_dotloop_diff.rs`, which runs
1167 /// a ROM corpus through BOTH paths and asserts identical framebuffer,
1168 /// palette-index framebuffer, audio, CPU-cycle count and full core
1169 /// snapshot **every frame** (so the fast path has never been unproven —
1170 /// only unshipped); and
1171 /// * **a clean-host Criterion confirmation** of the win: `full_frame`
1172 /// `nes_run_frame_nestest` 4.4343 ms -> 3.9331 ms, **-11.3%**, on a quiet
1173 /// host, reproducing A1's interleaved +12.3% measurement. The
1174 /// rendering-disabled `flowing_palette` workload is unchanged (-0.07%,
1175 /// noise) because its guard bails at `rendering_enabled()`.
1176 ///
1177 /// Promotion changes the *default*, not the behaviour: the frame the fast
1178 /// path produces is the frame the exact path produces, by construction and
1179 /// by test. `false` still selects the fully-general per-dot path and
1180 /// remains the fallback for any future doubt.
1181 pub(crate) fast_dotloop: bool,
1182
1183 /// Optional per-PPU-dot state trace (Session-10 observability
1184 /// tooling). Gated on the `ppu-state-trace` cargo feature so
1185 /// the default build pays no memory or codegen cost. See
1186 /// `docs/adr/0005-ppu-state-trace.md`.
1187 #[cfg(feature = "ppu-state-trace")]
1188 pub(crate) state_trace: Option<crate::state_trace::PpuStateTrace>,
1189
1190 /// The CPU cycle the dots being ticked belong to, stamped into every state
1191 /// record. Written once per CPU cycle by the bus, before those dots run.
1192 ///
1193 /// Feature-gated, so the default build carries neither the field nor the
1194 /// store: this is bookkeeping for a diagnostic, and the tick path is the
1195 /// hottest loop in the emulator.
1196 #[cfg(feature = "ppu-state-trace")]
1197 pub(crate) trace_cpu_cycle: u64,
1198 /// Per-dot PPU bus address capture. See [`crate::fetch_trace`].
1199 #[cfg(feature = "ppu-fetch-trace")]
1200 pub(crate) fetch_trace: Option<crate::fetch_trace::FetchTrace>,
1201
1202 /// v1.2.0 beta.2 (Workstream C3) — per-pixel HD-pack tile-source buffer
1203 /// (256 × 240 [`HdTileSource`] records), written in [`Self::emit_pixel`]
1204 /// in lockstep with [`Self::index_framebuffer`]. Output-only telemetry,
1205 /// gated on the `hd-pack` cargo feature so the default build pays no
1206 /// memory or codegen cost. Not part of the save-state.
1207 #[cfg(feature = "hd-pack")]
1208 pub(crate) hd_tile_source: Box<[HdTileSource]>,
1209
1210 /// v1.2.0 beta.2 (Workstream C3) — BG tile CHR base address latched at
1211 /// `fetch_bg_lo` time, then reloaded into the 2-stage `hd_bg_addr_*`
1212 /// queue in `reload_bg_shift_regs` so it tracks the BG pattern shift
1213 /// registers tile-for-tile. Pure telemetry; only touched when `hd-pack`
1214 /// is enabled.
1215 #[cfg(feature = "hd-pack")]
1216 pub(crate) hd_bg_addr_latch: u16,
1217 /// CHR base address of the BG tile currently feeding the shifters' high
1218 /// byte (the tile being displayed). See [`Self::hd_bg_addr_latch`].
1219 #[cfg(feature = "hd-pack")]
1220 pub(crate) hd_bg_addr_cur: u16,
1221 /// CHR base address of the next BG tile (shifters' low byte). Promoted to
1222 /// `hd_bg_addr_cur` on the prefetch byte-shift / per-tile boundary.
1223 #[cfg(feature = "hd-pack")]
1224 pub(crate) hd_bg_addr_next: u16,
1225 /// CHR base address fetched per sprite slot at `fetch_sprite_tile` time,
1226 /// consumed by `emit_pixel` for HD-pack sprite substitution.
1227 #[cfg(feature = "hd-pack")]
1228 pub(crate) hd_spr_addr: [u16; 8],
1229 /// The sprite's ORIGIN screen X per slot (the un-decremented `spr_x`), so
1230 /// `emit_pixel` can derive the column within the sprite for HD positioning.
1231 #[cfg(feature = "hd-pack")]
1232 pub(crate) hd_spr_x: [u8; 8],
1233 /// The sprite's flip-baked texel ROW (0..=7) per slot, captured at fetch (the
1234 /// `emit_pixel`-time scanline isn't enough to recover it post-shift).
1235 #[cfg(feature = "hd-pack")]
1236 pub(crate) hd_spr_off_y: [u8; 8],
1237 /// Absolute CHR-ROM tile index (`chr_phys/16`, or [`HD_CHR_RAM`]) tracked in
1238 /// lock-step with the `hd_bg_addr_*` cascade — the CHR-ROM HD-pack key.
1239 #[cfg(feature = "hd-pack")]
1240 pub(crate) hd_bg_idx_latch: u32,
1241 /// CHR-ROM tile index of the BG tile feeding the shifters' high byte.
1242 #[cfg(feature = "hd-pack")]
1243 pub(crate) hd_bg_idx_cur: u32,
1244 /// CHR-ROM tile index of the next BG tile (shifters' low byte).
1245 #[cfg(feature = "hd-pack")]
1246 pub(crate) hd_bg_idx_next: u32,
1247 /// Absolute CHR-ROM tile index per sprite slot (or [`HD_CHR_RAM`]).
1248 #[cfg(feature = "hd-pack")]
1249 pub(crate) hd_spr_idx: [u32; 8],
1250
1251 /// v2.3.2 "Lucid" — per-byte write attribution for CIRAM / OAM / palette RAM.
1252 ///
1253 /// `None` (the default) until the frontend arms it via
1254 /// [`Self::set_write_attribution`], so an unarmed `debug-hooks` build pays one
1255 /// `Option` test per PPU-memory write and no heap at all. Output-only
1256 /// telemetry: nothing in the emulation path reads it, so an armed store is
1257 /// bit-identical to an unarmed one. Deliberately NOT part of the save-state —
1258 /// attribution describes writes *this session* performed, and a restored
1259 /// state's bytes have no such history (see [`crate::provenance`]).
1260 #[cfg(feature = "debug-hooks")]
1261 pub(crate) write_attrib: Option<Box<crate::provenance::WriteAttribution>>,
1262 /// The `(pc, cycle)` of the CPU instruction currently executing, pushed down
1263 /// by the bus before each `$2000-$3FFF` register write and before each OAM
1264 /// DMA burst. Stamped into every attribution record made while it is set.
1265 ///
1266 /// The PPU cannot derive this itself: it never sees the CPU's program
1267 /// counter, and the effective VRAM destination the bus would need to record
1268 /// the attribution itself lives in the PPU's internal `v` register. Splitting
1269 /// the two halves this way is what lets each side contribute only what it
1270 /// actually knows.
1271 #[cfg(feature = "debug-hooks")]
1272 pub(crate) attrib_pc: u16,
1273 /// CPU cycle counterpart of [`Self::attrib_pc`].
1274 #[cfg(feature = "debug-hooks")]
1275 pub(crate) attrib_cycle: u64,
1276 /// The `(pc, cycle)` of the `STA $4014` that armed the in-flight OAM DMA.
1277 ///
1278 /// Separate from [`Self::attrib_pc`] because the burst does not run during
1279 /// the triggering instruction: `$4014` only sets `dma_pending`, and the 513
1280 /// or 514 DMA cycles are then stolen from the instructions that follow. By
1281 /// the time the first byte lands, [`Self::attrib_pc`] has already advanced to
1282 /// whichever instruction is being halted — an answer that is true about the
1283 /// *timing* and wrong about the *cause*. The bus latches this pair at the
1284 /// `$4014` write via [`Self::latch_dma_attrib_context`] so every byte of the
1285 /// burst names the store that actually caused it.
1286 #[cfg(feature = "debug-hooks")]
1287 pub(crate) dma_attrib_pc: u16,
1288 /// CPU cycle counterpart of [`Self::dma_attrib_pc`].
1289 #[cfg(feature = "debug-hooks")]
1290 pub(crate) dma_attrib_cycle: u64,
1291
1292 /// v2.3.2 "Lucid" phase 2 — per-pixel provenance for the current frame.
1293 ///
1294 /// `None` until armed, like [`Self::write_attrib`]. Overwritten in place
1295 /// every frame, exactly like the framebuffer it shadows.
1296 #[cfg(feature = "debug-hooks")]
1297 pub(crate) prov_frame: Option<Box<crate::provenance::PixelProvenanceFrame>>,
1298 /// Fast "is provenance armed?" flag, mirroring `prov_frame.is_some()`.
1299 ///
1300 /// `emit_pixel` runs 61,440 times a frame and is one of the two hottest
1301 /// functions in the emulator, so the per-pixel guard is a plain `bool` load
1302 /// rather than an `Option<Box<..>>` discriminant behind a pointer. Same
1303 /// shape as the bus's `event_logging` / `access_logging` flags.
1304 #[cfg(feature = "debug-hooks")]
1305 pub(crate) prov_armed: bool,
1306 /// Nametable address of the most recent NT fetch, awaiting commit.
1307 ///
1308 /// Held separately from [`Self::prov_bg_latch`] because the PPU performs two
1309 /// **dummy nametable fetches** at dots 337-340, after the pre-render line's
1310 /// last real tile has been fetched but before the visible line's first
1311 /// reload consumes it. Writing the NT address straight into the latch let
1312 /// those dummies overwrite the pending tile, so the first visible tile group
1313 /// reported the address of the tile after it. A tile is defined when its
1314 /// PATTERN is fetched — which the dummy fetches never do — so the pending
1315 /// addresses are committed in `fetch_bg_lo`.
1316 #[cfg(feature = "debug-hooks")]
1317 pub(crate) prov_nt_pending: u16,
1318 /// Attribute address awaiting commit. See [`Self::prov_nt_pending`].
1319 #[cfg(feature = "debug-hooks")]
1320 pub(crate) prov_at_pending: u16,
1321 /// Addresses of the background tile most recently FETCHED (the tile two
1322 /// slots ahead of the one on screen). Committed at pattern-fetch time;
1323 /// promoted through `next` into `cur` by the same shift-register reloads
1324 /// that move the pattern bytes, so the cascade stays tile-for-tile aligned
1325 /// with what the shifters are emitting.
1326 #[cfg(feature = "debug-hooks")]
1327 pub(crate) prov_bg_latch: ProvBgAddrs,
1328 /// Addresses of the background tile currently BEING DISPLAYED — the one
1329 /// `emit_pixel` must report.
1330 ///
1331 /// This cascade exists because `v` cannot answer the question: by the time a
1332 /// tile's pixels reach the screen, `v` has already advanced two tiles past
1333 /// it. Deriving the nametable address from `v` at emit time would be wrong
1334 /// for every pixel, and wrong in a way that looks plausible.
1335 #[cfg(feature = "debug-hooks")]
1336 pub(crate) prov_bg_cur: ProvBgAddrs,
1337 /// Addresses of the next background tile (the shifters' low byte), promoted
1338 /// into [`Self::prov_bg_cur`] on the per-tile boundary.
1339 #[cfg(feature = "debug-hooks")]
1340 pub(crate) prov_bg_next: ProvBgAddrs,
1341 /// Pattern address fetched per sprite slot, so a sprite pixel can report the
1342 /// CHR row behind it. Mirrors the `hd-pack` `hd_spr_addr` capture, kept
1343 /// separate so neither feature's telemetry depends on the other being on.
1344 #[cfg(feature = "debug-hooks")]
1345 pub(crate) prov_spr_addr: [u16; 8],
1346}
1347
1348/// The three VRAM addresses that produced one background tile.
1349///
1350/// Carried as a unit through the PPU's internal fetch → display cascade
1351/// (`latch` → `next` → `cur`), so a promotion is one struct copy instead of
1352/// three separate field moves that could drift out of step with each other.
1353#[cfg(feature = "debug-hooks")]
1354#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
1355pub struct ProvBgAddrs {
1356 /// Nametable address the tile number came from.
1357 pub nt: u16,
1358 /// Attribute address the palette group came from. Carried rather than
1359 /// derived, because an MMC5 vertical split supplies its own attribute
1360 /// address that the standard `$23C0 | ...` arithmetic cannot produce.
1361 pub at: u16,
1362 /// CHR address of the pattern row (the low-plane address; the high plane is
1363 /// `+8`).
1364 pub pattern: u16,
1365}
1366
1367/// v2.0 Phase 6 (`mc-ppu-subpos`): the analog `$2001` PPUMASK write delay.
1368///
1369/// In PPU dots. `TriCNES` applies a `$2001` write 2-3 dots after the CPU write
1370/// (`PPU_Update2001Delay`, sub-dot alignment dependent); this emulator's reload
1371/// sits one dot later than that plus the existing 1-dot render gate, so the
1372/// effective default is 4. Runtime-tunable (an atomic) so the exact phase can be
1373/// swept against the `BG Serial In` / `Stale BG Shift` keys without rebuilding.
1374/// The `mc-ppu-subpos` name in the lines above is the v2.0 Phase 6 WORKSTREAM,
1375/// not a cargo feature: no manifest declares one, and the read at the BG-reload
1376/// freeze below is unconditional. Kept as the historical label rather than
1377/// rewritten, but do not go looking for a flag to turn on.
1378pub static MASK_WRITE_DELAY: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(4);
1379
1380/// v2.6.18 DERIVATION KNOB: depth, in PPU dots, of the rendering-enable
1381/// pipeline (`rendering_enabled_delayed`).
1382///
1383/// 1 = shipped semantics exactly (a consumer sees the PREVIOUS dot's rendering
1384/// value). 0 = no delay, consumers see the live mask. 2 = two dots.
1385///
1386/// This delay and the write-commit point are ONE quantity split across two
1387/// places: the observable effect time is the commit offset PLUS this depth.
1388/// v2.6.17 swept this 1..4 and found 1 optimal -- but only at the SHIPPED commit
1389/// point. Under phi2 the commit is half a dot later, so the depth reproducing
1390/// the same effect time is one LESS, exactly as `mask_for_skip_check` needed
1391/// 2 -> 1. Sweeping at the shipped placement answers a different question.
1392///
1393/// Feature-gated; the shipped build has no atomic load on the per-dot path.
1394#[cfg(feature = "phi2-write-sweep")]
1395pub static RENDER_GATE_LAG: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(1);
1396
1397/// Whether the specialized dot paths may be taken this dot.
1398///
1399/// Their guards prove a ONE-dot rendering history (`rendering_enabled_delayed`
1400/// and `prev_rendering_enabled`), which is exactly what depths 0 and 1 read. At
1401/// depth >= 2 the gate reads [`Ppu::render_gate_prev2`] — rendering from two
1402/// dots ago — and the guard says nothing about it, so a dot with rendering
1403/// enabled for the last two dots but disabled before them would run the
1404/// rendering-enabled fast body while the general path gates it off. Excluding
1405/// the fast paths at depth >= 2 keeps one code path answering the sweep's
1406/// question instead of two that disagree.
1407// Dead under `ppu-state-trace` for the same reason `tick_visible_render_fast`
1408// is: that feature `cfg`s the fast-path dispatch out entirely, so the guard
1409// term this function supplies has no caller. Live in every other build --
1410// the one shape that earns an `allow` rather than a deletion, and scoped to
1411// that feature so it cannot suppress a real finding elsewhere.
1412#[cfg(feature = "phi2-write-sweep")]
1413#[cfg_attr(feature = "ppu-state-trace", allow(dead_code))]
1414#[inline]
1415fn fast_dot_paths_valid() -> bool {
1416 // The OAM2 pipeline is shifted only on the general path, and the fast
1417 // bodies read the gate, so an armed OAM2 knob must take the general path
1418 // for the same reason a depth-2 render gate must. Learned the expensive
1419 // way on #506: a bypassed pipeline does not merely go stale, it makes the
1420 // swept cell measure something that is not the configuration named.
1421 // `SCROLL_GATE_LAG` joins them for a THIRD instance of the same trap: the
1422 // visible fast body performs its own dot-256 `inc_vert_v()`, so an armed
1423 // scroll knob is bypassed entirely and every swept cell reads identical --
1424 // which is exactly what the first run of that sweep reported, four depths
1425 // all at 143/144 with nothing gained and nothing lost.
1426 RENDER_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) <= 1
1427 && OAM2_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) == 2
1428 && SCROLL_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) == u8::MAX
1429}
1430
1431/// Default build: the depth is the shipped 1, so the guards are sufficient and
1432/// this folds away at compile time — the fast paths are reached exactly as
1433/// before.
1434#[cfg(not(feature = "phi2-write-sweep"))]
1435#[inline]
1436const fn fast_dot_paths_valid() -> bool {
1437 true
1438}
1439
1440/// The default build's guard term must be a compile-time `true`, or adding it
1441/// would have changed which dots take the specialized paths — and those paths
1442/// exist to be byte-identical. Asserted at compile time rather than in a test,
1443/// because "it folds away" is a property of the constant, not of a run.
1444#[cfg(not(feature = "phi2-write-sweep"))]
1445const _: () = assert!(fast_dot_paths_valid());
1446/// v2.6.18 condition-2 DIAGNOSTIC: the dot on scanline 241 that sets the VBL
1447/// flag. Default 1, which is what nesdev specifies and what ships.
1448///
1449/// Exists to separate two explanations for the six NMI entries failing when the
1450/// CPU access moves to the cycle's last dot: a pure one-dot relative shift
1451/// between the `$2002` read and VBL-set, versus a defect in the read placement
1452/// itself. Moving this is NOT a fix and must never be shipped non-default.
1453#[cfg(feature = "phi2-write-sweep")]
1454pub static VBL_SET_DOT: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(1);
1455
1456/// v2.6.18 condition-2 DIAGNOSTIC: the depth of the odd-frame-skip gate's
1457/// `$2001` pipeline, in stages. Default 2, which is what ships.
1458///
1459/// The skip gate is the most explicitly-documented compensation in this PPU --
1460/// its own comment says lockstep applies the write at the START of a CPU cycle
1461/// while hardware latches at phi2 -- so its depth is coupled to where the write
1462/// access lands, and sweeping the two independently answers a different
1463/// question than sweeping them together.
1464#[cfg(feature = "phi2-write-sweep")]
1465pub static SKIP_GATE_LAG: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(2);
1466
1467/// v2.6.18: the depth of the dot-256 vertical increment's own `$2001` gate.
1468///
1469/// `u8::MAX` (the default) means "ride the shared `rendering_gate`", which is
1470/// what ships. This is the consumer `Frozen OAM2 Increment` turns on: the ROM
1471/// enables rendering ON dot 256 and states that the vertical scroll is NOT
1472/// incremented, and our enable lands a dot early.
1473#[cfg(feature = "phi2-write-sweep")]
1474pub static SCROLL_GATE_LAG: core::sync::atomic::AtomicU8 =
1475 core::sync::atomic::AtomicU8::new(u8::MAX);
1476
1477/// v2.6.18: the depth of the dot-339 sprite-counter re-arm's own `$2001` gate.
1478///
1479/// `u8::MAX` (the default) means "ride the shared `rendering_gate`", which is
1480/// what ships. Any other value gives the re-arm its own depth in the shared
1481/// history, which is the question `Stale Sprite Shift Regs` and
1482/// `Frozen OAM2 Increment` forced: at the shipped placement the frozen-flag
1483/// machinery closes only at `RENDER_GATE_LAG = 2`, and that same depth is what
1484/// breaks the re-arm, which wants 1. Two consumers, one gate, opposite
1485/// requirements -- the shape v2.6.5 resolved by separating the background
1486/// shifters' reload from their shift clock.
1487#[cfg(feature = "phi2-write-sweep")]
1488pub static SPRITE_REARM_LAG: core::sync::atomic::AtomicU8 =
1489 core::sync::atomic::AtomicU8::new(u8::MAX);
1490
1491/// v2.6.18: the depth of the OAM2 counter's `$2001` gate, in dots. Default 0 =
1492/// the live mask, which is what ships.
1493///
1494/// Every other rendering consumer here is delayed and this one is not, so the
1495/// two OAM2 catalog entries -- which share it -- can only be satisfied at
1496/// different CPU access placements. Sweeping the depth asks whether that is the
1497/// reason, which no placement sweep can. Depth 3 is what closes
1498/// `Frozen OAM2 Increment`; the borrowed two-stage pipeline it first used could
1499/// not reach that, which is why the field below is dedicated.
1500#[cfg(feature = "phi2-write-sweep")]
1501pub static OAM2_GATE_LAG: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(2);
1502
1503impl Ppu {
1504 /// New PPU in power-on state.
1505 #[must_use]
1506 // The power-on field initialization is naturally long (the PPU has many
1507 // state fields); the feature-gated `hd-pack` initializers nudge it over the
1508 // 100-line lint. Splitting the struct literal would hurt readability.
1509 #[allow(clippy::too_many_lines)]
1510 pub fn new(region: PpuRegion) -> Self {
1511 let mut p = Self {
1512 region,
1513 ctrl: PpuCtrl::empty(),
1514 mask: PpuMask::empty(),
1515 mask_for_skip_check: PpuMask::empty(),
1516 dot0_replaced: false,
1517 mask_skip_pipe1: PpuMask::empty(),
1518 status: PpuStatus::empty(),
1519 oam_addr: 0,
1520 data_buffer: 0,
1521 render_data_bus: 0,
1522 ppudata_sm_countdown: 0,
1523 ppudata_v_inc_pending: false,
1524 spr_fetch_lo_raw: [0; 8],
1525 spr_fetch_hi_raw: [0; 8],
1526 ppudata_spr0_nt_addr: 0x2000,
1527 octal_latch: 0,
1528 address_bus: 0,
1529 ale_armed: false,
1530 pattern_latch_stale: false,
1531 copy_v_delay: 0,
1532 v: 0,
1533 t: 0,
1534 x: 0,
1535 w: false,
1536 ciram: vec![0u8; 0x0800].into_boxed_slice(),
1537 oam: vec![0u8; 0x0100].into_boxed_slice(),
1538 secondary_oam: [0xFF; 32],
1539 oam_bus_copybuffer: 0xFF,
1540 oam_bus_secondary: [0xFF; 32],
1541 oam_bus_addr_h: 0,
1542 oam_bus_addr_l: 0,
1543 oam_bus_secondary_addr: 0,
1544 oam_bus_copy_done: false,
1545 oam_bus_sprite_in_range: false,
1546 oam_bus_overflow_counter: 0,
1547 palette_ram: [0u8; 32],
1548 // v2.1.4 F2.3 — OAM decay is off by default; the timestamps start at 0
1549 // (row "last touched at cycle 0"). While disabled they are never read,
1550 // and `set_oam_decay(true)` re-bases them to the current cycle so
1551 // enabling mid-run does not instantly decay every row.
1552 oam_decay_cycles: [0; 32],
1553 oam_decay_enabled: false,
1554 open_bus: 0,
1555 open_bus_decay: [0; 3],
1556 nmi_line: false,
1557 suppress_vbl_this_frame: false,
1558 last_a12_level: false,
1559 // Power-up position matches Mesen2's NesPpu::Reset(false) endpoint
1560 // (_scanline=-1, _cycle=340). After the first PPU tick, wraps to
1561 // (scanline=0, dot=0, frame+=1), putting the post-power-on PPU
1562 // position within ~2 dots of Mesen2's. Combined with the 8-cycle
1563 // CPU reset (see Cpu::reset), this closes the +344-dot PPU offset
1564 // identified empirically in Session-13 (docs/audit/
1565 // session-13-cpu-boot-fix-2026-05-21.md).
1566 //
1567 // SESSION-29 CRITICAL FINDING: Option (a) "PPU re-baseline"
1568 // empirically attempted and DOES NOT CLOSE THE C1 AXIS.
1569 // Shifting PPU init by +2 dots to (scanline=0, dot=1):
1570 // - Generates 24 snapshot regressions (audio_db, visual,
1571 // m22, Cascade A) — all are "expected" cosmetic shifts.
1572 // - BUT the cpu_interrupts_v2/{2,3,5}_strict probes STILL
1573 // FAIL — confirmed via `cargo test ... --include-ignored`.
1574 //
1575 // The +2 dot shift moves everything uniformly: VBL set position
1576 // AND BIT $2002 read position both shift by +2 dots, preserving
1577 // the relative race-window relationship. The BIT $2002 polling
1578 // loop inside blargg `sync_vbl` still hits the pre-VBL-set side
1579 // of the race window.
1580 //
1581 // CONCLUSION: closing C1 requires changing the PHASE
1582 // RELATIONSHIP between CPU and PPU (Option b — master-clock-
1583 // precise scheduling refactor), NOT a global PPU init shift.
1584 // The 4 C1 IRQ-timing residuals are deferred to v2.0 with the
1585 // master-clock refactor; v1.0.0 ships at 90.65% AccuracyCoin
1586 // with the 4 residuals documented as v2.0-deferred. See
1587 // `docs/audit/session-29-c1-axis-final-conclusion-2026-05-23.md`
1588 // + `docs/audit/session-29-option-a-empirical-falsification.md`.
1589 dot: 340,
1590 scanline: region.prerender_line(),
1591 frame: 0,
1592 frame_complete: false,
1593 post_reset_mask_remaining: region.post_reset_mask_cycles(),
1594 nt_latch: 0,
1595 at_latch: 0,
1596 bg_lo_latch: 0,
1597 bg_hi_latch: 0,
1598 bg_shift_lo: 0,
1599 bg_shift_hi: 0,
1600 at_shift_lo: 0,
1601 at_shift_hi: 0,
1602 ex_attr_latch: None,
1603 bg_split_latch: None,
1604 spr_shift_lo: [0; 8],
1605 spr_shift_hi: [0; 8],
1606 spr_attr: [0; 8],
1607 spr_x: [0; 8],
1608 spr_halted: [true; 8],
1609 spr_count: 0,
1610 spr_zero_in_line: false,
1611 sprite_eval_read_latch: 0xFF,
1612 sprite_eval_n: 0,
1613 sprite_eval_m: 0,
1614 sprite_eval_found: 0,
1615 sprite_eval_sec_idx: 0,
1616 sprite_eval_copying: false,
1617 sprite_eval_done: false,
1618 sprite_eval_overflow_search: false,
1619 sprite_eval_zero_found: false,
1620 sprite_eval_first_iter: false,
1621 oam_corruption_pending: false,
1622 oam_corruption_index: 0,
1623 oam_corruption_disabled: false,
1624 oam_corruption_disabled_instant: false,
1625 oam2_addr: 0,
1626 oam2_fetch_addr: 0,
1627 oam2_overflowed: false,
1628 oam2_fetch_frozen: false,
1629 prev_rendering_enabled: false,
1630 rendering_enabled_delayed: false,
1631 #[cfg(feature = "phi2-write-sweep")]
1632 render_gate_prev2: false,
1633 rendering_enabled_delayed2: false,
1634 spr_rearm_deferred: false,
1635 #[cfg(feature = "phi2-write-sweep")]
1636 sweep_mask_history: [PpuMask::empty(); 4],
1637 bg_reload_render: false,
1638 mask_write_delay: 0,
1639 cached_visible: false,
1640 cached_pre_render: false,
1641 cached_render_line: false,
1642 #[cfg(feature = "ppu-idle-line-fast")]
1643 cached_idle_line: false,
1644 flags_cached_scanline: i16::MIN,
1645 active_palette: crate::palette::PpuPalette::Composite2C02,
1646 rgba_lut: build_rgba_lut(crate::palette::PpuPalette::Composite2C02),
1647 custom_palette: None,
1648 is_2c05: false,
1649 id_2c05: 0,
1650 // v2.1.7 P5 — default revision models no extra corruption; default
1651 // power-up palette is all-zero. Both keep the default build
1652 // byte-identical (the `palette_ram: [0u8; 32]` above already reflects
1653 // the `PaletteInit::Zeroed` default).
1654 die_revision: PpuRevision::Rp2c02H,
1655 power_up_palette: PaletteInit::Zeroed,
1656 framebuffer: vec![0u8; FRAMEBUFFER_LEN].into_boxed_slice(),
1657 index_framebuffer: vec![0u16; FRAMEBUFFER_PIXELS].into_boxed_slice(),
1658 #[cfg(feature = "ppu-fetch-trace")]
1659 fast_path_hits: 0,
1660 dot_counter: 0,
1661 frame_ntsc_phase: 0,
1662 extra_scanlines: 0,
1663 sprite_limit_disabled: false,
1664 spr_extra_count: 0,
1665 spr_extra_lo: [0; MAX_EXTRA_SPRITES],
1666 spr_extra_hi: [0; MAX_EXTRA_SPRITES],
1667 spr_extra_attr: [0; MAX_EXTRA_SPRITES],
1668 spr_extra_x: [0; MAX_EXTRA_SPRITES],
1669 extra_lines_remaining: 0,
1670 // v2.2.3 performance pass: promoted to the default (was `false`
1671 // through v2.2.2). Byte-identical to the exact path by
1672 // construction and by `fast_dotloop_diff.rs`; -11.3% on the
1673 // rendering-heavy `full_frame` bench. See the field's rustdoc.
1674 fast_dotloop: true,
1675 #[cfg(feature = "ppu-state-trace")]
1676 state_trace: None,
1677 #[cfg(feature = "ppu-state-trace")]
1678 trace_cpu_cycle: 0,
1679 #[cfg(feature = "ppu-fetch-trace")]
1680 fetch_trace: None,
1681 #[cfg(feature = "hd-pack")]
1682 hd_tile_source: vec![HdTileSource::default(); FRAMEBUFFER_PIXELS].into_boxed_slice(),
1683 #[cfg(feature = "hd-pack")]
1684 hd_bg_addr_latch: HD_TILE_NONE,
1685 #[cfg(feature = "hd-pack")]
1686 hd_bg_addr_cur: HD_TILE_NONE,
1687 #[cfg(feature = "hd-pack")]
1688 hd_bg_addr_next: HD_TILE_NONE,
1689 #[cfg(feature = "hd-pack")]
1690 hd_spr_addr: [HD_TILE_NONE; 8],
1691 #[cfg(feature = "hd-pack")]
1692 hd_spr_x: [0; 8],
1693 #[cfg(feature = "hd-pack")]
1694 hd_spr_off_y: [0; 8],
1695 #[cfg(feature = "hd-pack")]
1696 hd_bg_idx_latch: HD_CHR_RAM,
1697 #[cfg(feature = "hd-pack")]
1698 hd_bg_idx_cur: HD_CHR_RAM,
1699 #[cfg(feature = "hd-pack")]
1700 hd_bg_idx_next: HD_CHR_RAM,
1701 #[cfg(feature = "hd-pack")]
1702 hd_spr_idx: [HD_CHR_RAM; 8],
1703 #[cfg(feature = "debug-hooks")]
1704 write_attrib: None,
1705 #[cfg(feature = "debug-hooks")]
1706 attrib_pc: 0,
1707 #[cfg(feature = "debug-hooks")]
1708 attrib_cycle: 0,
1709 #[cfg(feature = "debug-hooks")]
1710 dma_attrib_pc: 0,
1711 #[cfg(feature = "debug-hooks")]
1712 dma_attrib_cycle: 0,
1713 #[cfg(feature = "debug-hooks")]
1714 prov_frame: None,
1715 #[cfg(feature = "debug-hooks")]
1716 prov_armed: false,
1717 #[cfg(feature = "debug-hooks")]
1718 prov_nt_pending: 0,
1719 #[cfg(feature = "debug-hooks")]
1720 prov_at_pending: 0,
1721 #[cfg(feature = "debug-hooks")]
1722 prov_bg_latch: ProvBgAddrs::default(),
1723 #[cfg(feature = "debug-hooks")]
1724 prov_bg_cur: ProvBgAddrs::default(),
1725 #[cfg(feature = "debug-hooks")]
1726 prov_bg_next: ProvBgAddrs::default(),
1727 // The "no pattern" sentinel, not 0 — 0 is a legitimate CHR address.
1728 // Unreachable at emit time today (a sprite is only selected when
1729 // `spr_idx != 0`, which implies a real fetch), but the adjacent
1730 // `hd_spr_addr` uses its own sentinel for exactly this reason and a
1731 // future reader should not have to re-derive why 0 was safe.
1732 #[cfg(feature = "debug-hooks")]
1733 prov_spr_addr: [crate::provenance::PATTERN_ADDR_NONE; 8],
1734 };
1735 // Clear status flags that match power-on per nesdev wiki: VBL is
1736 // unspecified on power-on. We start clear.
1737 p.status = PpuStatus::empty();
1738 p
1739 }
1740
1741 /// Configure the PPU's hardware variant for Vs. System / PlayChoice-10
1742 /// arcade carts.
1743 ///
1744 /// `palette` selects the output palette (the RGB PPUs replace the 2C02
1745 /// composite palette with a fixed hardware RGB lookup). `is_2c05` enables
1746 /// the 2C05's register quirks: a write to `$2000` sets MASK and a write to
1747 /// `$2001` sets CTRL (swapped), and a `$2002` read ORs `id` into its low
1748 /// bits. For a 2C02 (the default NES/Famicom path) this is never called, so
1749 /// `active_palette` stays [`crate::palette::PpuPalette::Composite2C02`],
1750 /// `is_2c05` stays `false`, and behaviour is byte-for-byte unchanged.
1751 pub const fn set_palette(
1752 &mut self,
1753 palette: crate::palette::PpuPalette,
1754 is_2c05: bool,
1755 id: u8,
1756 ) {
1757 self.active_palette = palette;
1758 // v2.8.0 Phase 4 — keep the per-pixel RGBA lookup in sync with the active
1759 // palette (byte-identical by construction; see `rgba_lut`). A loaded `.pal`
1760 // (`custom_palette`) overrides the built-in table; `rebuild_rgba_lut`
1761 // honours it.
1762 self.rebuild_rgba_lut();
1763 self.is_2c05 = is_2c05;
1764 self.id_2c05 = id;
1765 }
1766
1767 /// v1.1.0 beta.1 (T-110-A3) — install (or clear with `None`) a custom 64-entry
1768 /// base palette from a loaded `.pal` file and rebuild the RGBA lookup. `None`
1769 /// restores the built-in palette for the active [`crate::palette::PpuPalette`]
1770 /// (byte-identical to default). A frontend presentation override.
1771 pub const fn set_custom_palette(&mut self, base: Option<[[u8; 3]; 64]>) {
1772 self.custom_palette = base;
1773 self.rebuild_rgba_lut();
1774 }
1775
1776 /// v3.1.0 (`T-SPRITE-LIMIT`) — draw the sprites beyond the eighth on a
1777 /// scanline (`true`), or not (`false`, the default and the hardware). See
1778 /// the field for what stays exact. Turning it off drops any extras already
1779 /// fetched, so the next scanline draws exactly eight.
1780 pub const fn set_sprite_limit_disabled(&mut self, disabled: bool) {
1781 self.sprite_limit_disabled = disabled;
1782 if !disabled {
1783 self.spr_extra_count = 0;
1784 }
1785 }
1786
1787 /// v3.1.0 — whether the sprites beyond the eighth are drawn.
1788 #[must_use]
1789 pub const fn sprite_limit_disabled(&self) -> bool {
1790 self.sprite_limit_disabled
1791 }
1792
1793 /// v3.1.0 — how many sprites beyond the eighth are fetched for the next
1794 /// scanline (`0` unless [`Self::sprite_limit_disabled`]).
1795 #[must_use]
1796 pub const fn extra_sprite_count(&self) -> u8 {
1797 self.spr_extra_count
1798 }
1799
1800 /// v1.7.0 "Forge" Workstream F3 — set the number of EXTRA blank vblank
1801 /// scanlines to insert per frame (the PPU extra-scanlines overclock).
1802 ///
1803 /// `0` (the default) is stock NES timing and is **byte-identical** to a PPU
1804 /// that never calls this. A non-zero value lengthens vblank by that many
1805 /// idle scanlines each frame (more CPU run-time, no visible change), at the
1806 /// existing dot resolution. Off by default; a frontend config knob, not part
1807 /// of the save-state. Distinct from the CPU-multiplier overclock (v3.1.0).
1808 ///
1809 /// Changing the count cancels any in-flight insertion for the current
1810 /// frame: the per-frame countdown (`extra_lines_remaining`) is
1811 /// reset to `0` so it cannot remain stale or out-of-bounds relative to
1812 /// the new `lines` (e.g. shrinking 8 → 2, or disabling N → 0). The next
1813 /// frame reloads the countdown from the new value at the insertion point.
1814 pub const fn set_extra_scanlines(&mut self, lines: u16) {
1815 self.extra_scanlines = lines;
1816 self.extra_lines_remaining = 0;
1817 }
1818
1819 /// v1.7.0 F3 — the currently-configured extra-scanline count (`0` = stock).
1820 #[must_use]
1821 pub const fn extra_scanlines(&self) -> u16 {
1822 self.extra_scanlines
1823 }
1824
1825 /// v2.1.8 A1 — enable/disable the specialized visible-scanline fast dot
1826 /// path. **Default ON since the v2.2.3 performance pass** (was OFF through
1827 /// v2.2.2); either setting produces the identical frame, so this selects a
1828 /// code path, not a behaviour. See [`Self::fast_dotloop`] and
1829 /// `docs/performance.md`.
1830 pub const fn set_fast_dotloop(&mut self, enabled: bool) {
1831 self.fast_dotloop = enabled;
1832 }
1833
1834 /// v2.1.8 A1 — whether the visible-scanline fast dot path is enabled.
1835 #[must_use]
1836 pub const fn fast_dotloop(&self) -> bool {
1837 self.fast_dotloop
1838 }
1839
1840 /// v1.1.0 beta.1 — the custom 64-entry base palette installed by
1841 /// [`Self::set_custom_palette`], or `None` for the built-in one.
1842 #[must_use]
1843 pub const fn custom_palette(&self) -> Option<[[u8; 3]; 64]> {
1844 self.custom_palette
1845 }
1846
1847 /// v2.9.8 — carry the host's settings from `prev` onto this freshly built
1848 /// PPU, so a power cycle (which rebuilds the PPU from [`Self::new`]) keeps
1849 /// them.
1850 ///
1851 /// # Why this lives on the PPU
1852 ///
1853 /// A power cycle is a cold boot of the console, not of the user's
1854 /// configuration. Until v2.9.8 the bus rebuilt the PPU and re-applied only
1855 /// the settings it also stored itself (the die revision, the power-up
1856 /// palette, the Vs. RGB palette), so every setting held ONLY here --
1857 /// the custom / generated palette, the extra-scanlines overclock, the fast
1858 /// dot path selector and the OAM-decay model -- silently reverted to its
1859 /// default, and each host had to remember to push it again. The list of
1860 /// what a setting is belongs next to the fields, so a new one is carried
1861 /// where it is declared; `snapshot_schema_audit.rs` cross-checks it against
1862 /// the fields that audit classifies as configuration.
1863 ///
1864 /// # What is carried, and what is not
1865 ///
1866 /// Carried: [`Self::custom_palette`] (the lookup table is rebuilt to
1867 /// honour it), [`Self::extra_scanlines`], [`Self::sprite_limit_disabled`],
1868 /// [`Self::fast_dotloop`] and
1869 /// [`Self::oam_decay_enabled`]. The decay switch goes through
1870 /// [`Self::set_oam_decay`], exactly as a host enabling it on a fresh
1871 /// console would, so the result is what a fresh boot with the setting
1872 /// applied produces.
1873 ///
1874 /// Not carried here: the die revision and the power-up palette are stored
1875 /// on the bus, which re-applies them (the power-up palette is a power-on
1876 /// FILL and must be rewritten, not copied); the active palette and the
1877 /// 2C05 identity are board identity, re-derived from the cartridge by the
1878 /// bus. The state and fetch traces are capture buffers, not settings: a
1879 /// cold boot ends the history they describe, as it does for the
1880 /// provenance stores, which the core moves across separately (armed, then
1881 /// emptied).
1882 ///
1883 /// Every carried value is a selector or an output override, so with
1884 /// every setting at its default this leaves the PPU byte-identical to
1885 /// [`Self::new`].
1886 pub const fn adopt_settings_from(&mut self, prev: &Self) {
1887 self.custom_palette = prev.custom_palette;
1888 self.rebuild_rgba_lut();
1889 self.set_extra_scanlines(prev.extra_scanlines);
1890 self.sprite_limit_disabled = prev.sprite_limit_disabled;
1891 self.fast_dotloop = prev.fast_dotloop;
1892 self.set_oam_decay(prev.oam_decay_enabled);
1893 }
1894
1895 /// v2.1.4 F2.3 — enable or disable the optional OAM-decay accuracy model.
1896 ///
1897 /// **Off by default.** When off (the default) OAM reads/writes never consult
1898 /// the decay state and the deterministic output is **byte-identical** to a
1899 /// build without the feature — `AccuracyCoin`, the commercial oracle, and the
1900 /// visual/`external_real_games` regression suites are unaffected. When on, the
1901 /// PPU refreshes each 8-byte OAM row on every read (sprite evaluation + `$2004`)
1902 /// and write; a row that goes un-refreshed for more than `OAM_DECAY_CPU_CYCLES`
1903 /// (3000) CPU cycles decays to Mesen2's canonical garbage pattern on the next
1904 /// read (`oam_decay_on_read`). The model is NTSC/Dendy-only (PAL's refresh
1905 /// cadence masks decay) — the region gate lives in the hooks, so it is safe to
1906 /// enable on any region.
1907 ///
1908 /// A frontend/config knob (re-applied on load like `region` / `active_palette`),
1909 /// **not** part of the save-state. Turning the model ON re-bases every row's
1910 /// timestamp to the current CPU cycle so a freshly-enabled model does not report
1911 /// every row as instantly decayed; turning it OFF leaves the timestamps as-is
1912 /// (they are simply no longer consulted).
1913 pub const fn set_oam_decay(&mut self, enabled: bool) {
1914 if enabled && !self.oam_decay_enabled {
1915 // Freshly enabling: treat every row as just-refreshed so the first
1916 // post-enable reads don't spuriously report a multi-second-old row as
1917 // decayed. Idempotent re-enables (already on) skip this so a long
1918 // rendering-disabled span already in progress keeps decaying.
1919 let now = self.dot_counter / 3;
1920 let mut i = 0;
1921 while i < self.oam_decay_cycles.len() {
1922 self.oam_decay_cycles[i] = now;
1923 i += 1;
1924 }
1925 }
1926 self.oam_decay_enabled = enabled;
1927 }
1928
1929 /// v2.1.4 F2.3 — whether the optional OAM-decay model is currently enabled.
1930 #[must_use]
1931 pub const fn oam_decay_enabled(&self) -> bool {
1932 self.oam_decay_enabled
1933 }
1934
1935 /// v2.1.7 P5 — select the emulated 2C02 die revision (see [`PpuRevision`]).
1936 ///
1937 /// The [`PpuRevision::default`] ([`PpuRevision::Rp2c02H`]) models no extra
1938 /// quirks, so at the default this is behaviorally inert and the PPU is
1939 /// byte-identical to a build without the field. Selecting
1940 /// [`PpuRevision::Rp2c02G`] additionally arms the OAMADDR (`$2003`)
1941 /// write-during-rendering OAM corruption glitch. A construction/config knob,
1942 /// re-applied on load like the region / active palette — not part of the
1943 /// save-state.
1944 pub const fn set_revision(&mut self, revision: PpuRevision) {
1945 self.die_revision = revision;
1946 }
1947
1948 /// v2.1.7 P5 — the currently-selected 2C02 die revision.
1949 #[must_use]
1950 pub const fn revision(&self) -> PpuRevision {
1951 self.die_revision
1952 }
1953
1954 /// v2.9.8 — CPU cycles left in the post-reset warm-up window, during which
1955 /// writes to `$2000`/`$2001`/`$2005`/`$2006` are ignored (`0` once the
1956 /// window has passed). Read-only; see [`PpuRegion::post_reset_mask_cycles`]
1957 /// and `docs/ppu-2c02.md` (§Power-up and reset).
1958 #[must_use]
1959 pub const fn warmup_cycles_remaining(&self) -> u32 {
1960 self.post_reset_mask_remaining
1961 }
1962
1963 /// v2.9.8 — end the post-reset warm-up window now, so `$2000`/`$2001`/
1964 /// `$2005`/`$2006` writes take effect from the next CPU cycle.
1965 ///
1966 /// This is how the opt-in Famicom console model is expressed at the PPU:
1967 /// the `NESdev` wiki's "PPU power up state" (§Famicom) documents that the
1968 /// Famicom ties the PPU's `/RESET` to 5 V while the CPU's `/RESET` rides a
1969 /// 0.47 µF
1970 /// capacitor, so at power-on the PPU begins initialising roughly one frame
1971 /// (about 29,781 CPU cycles) before the CPU leaves reset. The warm-up window
1972 /// is 29,658 cycles on NTSC, shorter than that frame, so by the time the
1973 /// CPU executes its first instruction the window has already closed. The
1974 /// PPU itself is unchanged; the console decides when its reset is released,
1975 /// which is why the caller (the console model in `rustynes-core`) owns the
1976 /// decision and this is only the mechanism.
1977 ///
1978 /// Only the window is touched. Every register and the frame position stay
1979 /// as they are, and the field it clears is already part of the save-state,
1980 /// so no snapshot change follows from calling it.
1981 pub const fn end_warmup(&mut self) {
1982 self.post_reset_mask_remaining = 0;
1983 }
1984
1985 /// v2.1.7 P5 — apply a power-up palette-RAM pattern (see [`PaletteInit`]).
1986 ///
1987 /// Writes all 32 palette-RAM bytes to the selected pattern and records the
1988 /// selection so a subsequent power-cycle can re-apply it. The
1989 /// [`PaletteInit::default`] ([`PaletteInit::Zeroed`]) writes all-zero — the
1990 /// established power-up state — so at the default this leaves the PPU
1991 /// byte-identical. Intended to be called at construction / power-on (palette
1992 /// RAM is not cleared on a warm reset, matching real hardware). It writes
1993 /// [`Self::palette_ram`] directly, which the snapshot already serializes, so
1994 /// no snapshot-format change is required.
1995 pub const fn apply_power_up_palette(&mut self, init: PaletteInit) {
1996 self.power_up_palette = init;
1997 match init {
1998 PaletteInit::Zeroed => {
1999 let mut i = 0;
2000 while i < self.palette_ram.len() {
2001 self.palette_ram[i] = 0;
2002 i += 1;
2003 }
2004 }
2005 PaletteInit::Blargg => {
2006 let mut i = 0;
2007 while i < self.palette_ram.len() {
2008 // Palette-RAM cells are 6-bit; mask to match a `$2007` write.
2009 self.palette_ram[i] = BLARGG_POWER_UP_PALETTE[i] & 0x3F;
2010 i += 1;
2011 }
2012 }
2013 }
2014 }
2015
2016 /// v2.1.7 P5 — the currently-selected power-up palette pattern.
2017 #[must_use]
2018 pub const fn power_up_palette(&self) -> PaletteInit {
2019 self.power_up_palette
2020 }
2021
2022 /// v2.1.4 F2.3 — `true` when the OAM-decay model should act this access:
2023 /// enabled AND the region is NTSC/Dendy (PAL's frequent refresh masks decay,
2024 /// so Mesen2 never decays there). This is the single gate every decay hook
2025 /// funnels through; at the default (disabled) it is a single bool test and the
2026 /// hooks are behaviour-neutral.
2027 #[inline]
2028 const fn oam_decay_active(&self) -> bool {
2029 self.oam_decay_enabled && !matches!(self.region, PpuRegion::Pal)
2030 }
2031
2032 /// v2.1.4 F2.3 — OAM-read decay hook. Call **immediately before** reading
2033 /// `oam[addr]` at every primary-OAM read site (the `$2004` read and both
2034 /// sprite-evaluation read paths). Implements the `NESdev`-documented OAM DRAM
2035 /// decay-on-read behavior (`NESdev` wiki "PPU OAM" — sprite RAM is dynamic and
2036 /// its cells decay; a read recharges the touched row):
2037 ///
2038 /// - If the model is inactive (disabled or PAL), this is a no-op — `oam` and
2039 /// the timestamps are left untouched, so the read is byte-identical to stock.
2040 /// - Else, for the 8-byte row containing `addr`: if the last touch was within
2041 /// [`OAM_DECAY_CPU_CYCLES`] CPU cycles, refresh the row's timestamp (the DRAM
2042 /// cell was recharged by this access). Otherwise the row has decayed — rewrite
2043 /// all 8 of its bytes to the canonical pattern `((sprAddr & 3) == 2) ?
2044 /// (sprAddr & 0xE3) : sprAddr` (the attribute byte keeps only its implemented
2045 /// bits; the others read back their own low address) and leave the stale
2046 /// timestamp (so the row keeps reading decayed until a write refreshes it,
2047 /// matching the documented decay behavior).
2048 ///
2049 /// The subsequent `oam[addr]` read then returns the (possibly decayed) byte.
2050 #[inline]
2051 fn oam_decay_on_read(&mut self, addr: u8) {
2052 if !self.oam_decay_active() {
2053 return;
2054 }
2055 let row = (addr >> 3) as usize;
2056 let now = self.dot_counter / 3;
2057 // Saturating (wrapping) subtraction: `now` is monotone ≥ the stored
2058 // timestamp in practice, but `wrapping_sub` keeps this total even across a
2059 // (astronomically unlikely) u64 counter wrap.
2060 let elapsed = now.wrapping_sub(self.oam_decay_cycles[row]);
2061 if elapsed <= OAM_DECAY_CPU_CYCLES {
2062 self.oam_decay_cycles[row] = now;
2063 } else {
2064 let base = addr & 0xF8;
2065 for i in 0..8u8 {
2066 let spr_addr = base | i;
2067 self.oam[spr_addr as usize] = if spr_addr & 0x03 == 0x02 {
2068 spr_addr & 0xE3
2069 } else {
2070 spr_addr
2071 };
2072 }
2073 }
2074 }
2075
2076 /// v2.1.4 F2.3 — OAM-write decay hook. Call **after** writing `oam[addr]` at
2077 /// every primary-OAM write site (`$2004` / OAM DMA). Implements the documented
2078 /// OAM DRAM decay-on-write refresh (`NESdev` wiki "PPU OAM"): a write recharges
2079 /// the row's DRAM cells, so refresh the
2080 /// row's last-touch timestamp. Inactive (disabled or PAL) ⇒ no-op, so the write
2081 /// path is byte-identical to stock at the default.
2082 #[inline]
2083 const fn oam_decay_on_write(&mut self, addr: u8) {
2084 if !self.oam_decay_active() {
2085 return;
2086 }
2087 self.oam_decay_cycles[(addr >> 3) as usize] = self.dot_counter / 3;
2088 }
2089
2090 /// Rebuild [`Self::rgba_lut`] from the custom palette when one is loaded,
2091 /// otherwise from the active built-in [`crate::palette::PpuPalette`].
2092 const fn rebuild_rgba_lut(&mut self) {
2093 self.rgba_lut = match &self.custom_palette {
2094 Some(base) => build_rgba_lut_from_base(base),
2095 None => build_rgba_lut(self.active_palette),
2096 };
2097 }
2098
2099 /// Map a CPU-visible PPU register index (0-7) to the internal register,
2100 /// applying the 2C05 `$2000`<->`$2001` swap.
2101 ///
2102 /// On a 2C05 a write/read of `$2000` (reg 0) targets MASK (reg 1) and vice
2103 /// versa; all other registers are unaffected. On every other PPU (the
2104 /// default path) this is the identity, so normal NES behaviour is unchanged.
2105 const fn map_register(&self, reg: u8) -> u8 {
2106 if self.is_2c05 {
2107 match reg & 7 {
2108 0 => 1,
2109 1 => 0,
2110 other => other,
2111 }
2112 } else {
2113 reg & 7
2114 }
2115 }
2116
2117 /// Returns a reference to the internal CIRAM (nametables).
2118 pub fn vram_ref(&self) -> &[u8] {
2119 &self.ciram
2120 }
2121
2122 /// Returns a mutable reference to the internal CIRAM (nametables).
2123 pub fn vram_mut(&mut self) -> &mut [u8] {
2124 &mut self.ciram
2125 }
2126
2127 /// Performs a soft-reset of the PPU (warm boot). Per `docs/ppu-2c02.md`:
2128 /// - PPUCTRL := 0
2129 /// - PPUMASK := 0
2130 /// - w toggle := 0
2131 /// - PPUSTATUS bits 7 (VBL) unchanged on real hardware (we leave it
2132 /// as-is for parity with `$2002`-race tests)
2133 /// - PPUDATA buffer := 0
2134 /// - Mask window restarts (writes to $2000/$2001/$2005/$2006 ignored
2135 /// for the documented number of cycles after reset).
2136 pub const fn reset(&mut self) {
2137 self.ctrl = PpuCtrl::empty();
2138 self.mask = PpuMask::empty();
2139 self.mask_for_skip_check = PpuMask::empty();
2140 self.mask_skip_pipe1 = PpuMask::empty();
2141 self.prev_rendering_enabled = false;
2142 self.rendering_enabled_delayed = false;
2143 self.rendering_enabled_delayed2 = false;
2144 self.spr_rearm_deferred = false;
2145 // The rendering-gate pipeline is the same class of state as the two
2146 // skip-check stages above and was simply missed. Reset preserves the
2147 // current dot, so a reset at dot 254 otherwise reaches dot 256 with
2148 // stale history and fires `inc_vert_v()` against an empty PPUMASK.
2149 // Found in review on #515; the two-dot stage widened the window that
2150 // made it observable, but `rendering_enabled_delayed` had the same hole.
2151 self.w = false;
2152 self.data_buffer = 0;
2153 self.post_reset_mask_remaining = self.region.post_reset_mask_cycles();
2154 self.nmi_line = false;
2155 // v2.1.4 F2.3 — mark every OAM row freshly refreshed at reset, matching
2156 // Mesen2's `NesPpu::Reset` (which stamps `_oamDecayCycles` with the current
2157 // clock unconditionally). Doing this regardless of the enable flag keeps the
2158 // timestamps sane if decay is toggled on after a reset, and is inert while
2159 // decay is off (the array is never read). `dot_counter / 3` is the current
2160 // CPU cycle (NTSC/Dendy have 3 dots per CPU cycle).
2161 let now = self.dot_counter / 3;
2162 let mut i = 0;
2163 while i < self.oam_decay_cycles.len() {
2164 self.oam_decay_cycles[i] = now;
2165 i += 1;
2166 }
2167 }
2168
2169 /// Returns `true` if the PPU is asserting the NMI line.
2170 #[must_use]
2171 pub const fn nmi_line(&self) -> bool {
2172 self.nmi_line
2173 }
2174
2175 /// Consume and return the per-frame "frame complete" latch.
2176 pub const fn take_frame_complete(&mut self) -> bool {
2177 let r = self.frame_complete;
2178 self.frame_complete = false;
2179 r
2180 }
2181
2182 /// Install a state-trace buffer. Subsequent calls to
2183 /// [`Self::tick`] will append one [`PpuStateRecord`] per dot
2184 /// for dots inside the buffer's filter window. Pre-existing
2185 /// records (if any) are dropped.
2186 ///
2187 /// Read-only: every call to [`Self::tick`] reads PPU state
2188 /// after the dot's effects have applied; it never mutates
2189 /// emulator state, so the determinism contract is preserved
2190 /// (`docs/architecture.md` §Determinism).
2191 ///
2192 /// See `docs/adr/0005-ppu-state-trace.md` and the rustdoc on
2193 /// [`crate::state_trace`].
2194 ///
2195 /// [`PpuStateRecord`]: crate::state_trace::PpuStateRecord
2196 #[cfg(feature = "ppu-state-trace")]
2197 pub fn enable_state_trace(&mut self, trace: crate::state_trace::PpuStateTrace) {
2198 self.state_trace = Some(trace);
2199 }
2200
2201 /// Stamp the CPU cycle that the dots ticked next belong to.
2202 ///
2203 /// Called by the bus at the START of each CPU cycle, so a record carries
2204 /// the number of the cycle it is part of rather than the following one —
2205 /// the off-by-one a co-simulation probe made on exactly this question, and
2206 /// which produced a finding that had to be retracted.
2207 #[cfg(feature = "ppu-state-trace")]
2208 pub const fn set_trace_cpu_cycle(&mut self, cycle: u64) {
2209 self.trace_cpu_cycle = cycle;
2210 }
2211
2212 /// Install a per-dot PPU bus address capture.
2213 ///
2214 /// The address bus is pin-observable, which is what makes it usable as a
2215 /// gate by an independent reimplementation; see [`crate::fetch_trace`] for
2216 /// why the address is captured rather than derived.
2217 #[cfg(feature = "ppu-fetch-trace")]
2218 pub fn enable_fetch_trace(&mut self, trace: crate::fetch_trace::FetchTrace) {
2219 self.fetch_trace = Some(trace);
2220 }
2221
2222 /// Remove and return the fetch trace, if one was installed.
2223 #[cfg(feature = "ppu-fetch-trace")]
2224 pub const fn take_fetch_trace(&mut self) -> Option<crate::fetch_trace::FetchTrace> {
2225 self.fetch_trace.take()
2226 }
2227
2228 /// Take the accumulated state trace, leaving the PPU's trace
2229 /// slot empty. Returns `None` if tracing was never enabled.
2230 #[cfg(feature = "ppu-state-trace")]
2231 #[must_use]
2232 pub const fn take_state_trace(&mut self) -> Option<crate::state_trace::PpuStateTrace> {
2233 self.state_trace.take()
2234 }
2235
2236 /// Borrow the in-flight state trace without taking it.
2237 #[cfg(feature = "ppu-state-trace")]
2238 #[must_use]
2239 pub const fn state_trace(&self) -> Option<&crate::state_trace::PpuStateTrace> {
2240 self.state_trace.as_ref()
2241 }
2242
2243 /// Build a [`PpuStateRecord`] snapshot from the PPU's
2244 /// current state. Used by the per-dot recording hook at the
2245 /// end of [`Self::tick`]; exposed publicly so external
2246 /// tooling (e.g. the trace fixture's end-of-frame snapshot)
2247 /// can re-use the canonical packer.
2248 ///
2249 /// [`PpuStateRecord`]: crate::state_trace::PpuStateRecord
2250 #[cfg(feature = "ppu-state-trace")]
2251 #[must_use]
2252 pub fn build_state_record(&self) -> crate::state_trace::PpuStateRecord {
2253 crate::state_trace::PpuStateRecord {
2254 // Frames easily exceed u16 over a 600-frame test run.
2255 // The `as u32` truncates the upper bits of the u64
2256 // counter — which is fine for any realistic capture
2257 // window (u32::MAX ≈ 71 days of NES wall time).
2258 frame: self.frame as u32,
2259 scanline: self.scanline,
2260 dot: self.dot,
2261 ctrl: self.ctrl.bits(),
2262 mask: self.mask.bits(),
2263 status: self.status.bits(),
2264 oam_addr: self.oam_addr,
2265 v: self.v,
2266 t: self.t,
2267 fine_x: self.x,
2268 w_toggle: self.w,
2269 sprite_eval_n: self.sprite_eval_n,
2270 sprite_eval_m: self.sprite_eval_m,
2271 sprite_eval_found: self.sprite_eval_found,
2272 sprite_eval_sec_idx: self.sprite_eval_sec_idx,
2273 sprite_eval_copying: self.sprite_eval_copying,
2274 sprite_eval_overflow_search: self.sprite_eval_overflow_search,
2275 sprite_eval_done: self.sprite_eval_done,
2276 sprite_eval_read_latch: self.sprite_eval_read_latch,
2277 spr_count: self.spr_count,
2278 spr_zero_in_line: self.spr_zero_in_line,
2279 spr_shift_lo: self.spr_shift_lo,
2280 spr_shift_hi: self.spr_shift_hi,
2281 spr_attr: self.spr_attr,
2282 spr_x: self.spr_x,
2283 bg_shift_lo: self.bg_shift_lo,
2284 bg_shift_hi: self.bg_shift_hi,
2285 at_shift_lo: self.at_shift_lo,
2286 at_shift_hi: self.at_shift_hi,
2287 nt_latch: self.nt_latch,
2288 at_latch: self.at_latch,
2289 bg_lo_latch: self.bg_lo_latch,
2290 bg_hi_latch: self.bg_hi_latch,
2291 secondary_oam: self.secondary_oam,
2292 oam_fnv1a64: crate::state_trace::fnv1a64(&self.oam),
2293 nmi_line: self.nmi_line,
2294 oam_bus_copybuffer: self.oam_data_bus_observed(),
2295 data_buffer: self.data_buffer,
2296 cpu_cycle: self.trace_cpu_cycle,
2297 }
2298 }
2299
2300 /// Borrow the (possibly partial) framebuffer.
2301 #[must_use]
2302 pub fn framebuffer(&self) -> &[u8] {
2303 &self.framebuffer
2304 }
2305
2306 /// v1.7.0 "Forge" Workstream B (B3) — overwrite the RGBA8 output framebuffer
2307 /// (the Lua `emu:setScreenBuffer(t)` paints output only). Copies up to the
2308 /// framebuffer length; a short source leaves the tail untouched. Output-only
2309 /// — it touches only the display buffer the frontend presents, NOT any
2310 /// register / latch / scroll state, so the determinism contract is
2311 /// unaffected (a later real frame fully repaints it). `debug-hooks`-gated and
2312 /// reached only through the script crate's gated post-frame path, so the
2313 /// shipped build is byte-identical.
2314 #[cfg(feature = "debug-hooks")]
2315 pub fn debug_set_framebuffer(&mut self, rgba: &[u8]) {
2316 let n = rgba.len().min(self.framebuffer.len());
2317 self.framebuffer[..n].copy_from_slice(&rgba[..n]);
2318 }
2319
2320 /// Borrow the parallel per-pixel **palette-index** framebuffer
2321 /// (256 × 240 `u16`s, each `(emphasis << 6) | colour`, 0..=511) used by the
2322 /// true composite `NES_NTSC` filter (T-110-A1). A faithful index-space mirror
2323 /// of [`Self::framebuffer`]; output-only, so the determinism contract holds.
2324 #[must_use]
2325 pub fn index_framebuffer(&self) -> &[u16] {
2326 &self.index_framebuffer
2327 }
2328
2329 /// Pack a background tile's active palette into Mesen's `PaletteColors` key
2330 /// form (`pr[base+3] | pr[base+2]<<8 | pr[base+1]<<16 | pr[0]<<24`, with
2331 /// `base = $3F00 | group<<2` and `pr[0]` the universal backdrop). Used only
2332 /// to key HD-pack tile replacements; output-only.
2333 #[cfg(feature = "hd-pack")]
2334 fn hd_bg_palette_colors(&self, group: u8) -> u32 {
2335 let base = 0x3F00 | (u16::from(group) << 2);
2336 let p0 = u32::from(self.read_palette(0x3F00) & 0x3F);
2337 let p1 = u32::from(self.read_palette(base | 1) & 0x3F);
2338 let p2 = u32::from(self.read_palette(base | 2) & 0x3F);
2339 let p3 = u32::from(self.read_palette(base | 3) & 0x3F);
2340 p3 | (p2 << 8) | (p1 << 16) | (p0 << 24)
2341 }
2342
2343 /// Pack a sprite tile's palette (`0xFF000000 | pr[base+3] | pr[base+2]<<8 |
2344 /// pr[base+1]<<16`, `base = $3F10 | group<<2`; the `0xFF` top byte is the
2345 /// sprite/BG discriminator and there is no `pr[0]` term).
2346 #[cfg(feature = "hd-pack")]
2347 fn hd_sprite_palette_colors(&self, group: u8) -> u32 {
2348 let base = 0x3F10 | (u16::from(group) << 2);
2349 let p1 = u32::from(self.read_palette(base | 1) & 0x3F);
2350 let p2 = u32::from(self.read_palette(base | 2) & 0x3F);
2351 let p3 = u32::from(self.read_palette(base | 3) & 0x3F);
2352 0xFF00_0000 | p3 | (p2 << 8) | (p1 << 16)
2353 }
2354
2355 /// v1.2.0 beta.2 (Workstream C3) — borrow the per-pixel HD-pack
2356 /// tile-source buffer (256 × 240 [`HdTileSource`] records, parallel to
2357 /// [`Self::index_framebuffer`]). Each entry names the CHR tile that
2358 /// produced the pixel. Output-only telemetry; the determinism /
2359 /// `AccuracyCoin` contract is unaffected. See
2360 /// `docs/ppu-2c02.md` §HD-pack tile-source export.
2361 #[cfg(feature = "hd-pack")]
2362 #[must_use]
2363 pub fn hd_tile_source(&self) -> &[HdTileSource] {
2364 &self.hd_tile_source
2365 }
2366
2367 /// The per-frame NTSC composite colour phase — the `videoPhase` the
2368 /// `NES_NTSC` filter feeds its signal generator. `0..=2` on NTSC; on
2369 /// PAL/Dendy it is the frame parity (`0..=1`). Snapshotted at the last frame
2370 /// boundary. Cosmetic (drives only the optional filter's dot-crawl).
2371 #[must_use]
2372 pub const fn ntsc_phase(&self) -> u8 {
2373 self.frame_ntsc_phase
2374 }
2375
2376 /// Snapshot the per-frame NTSC colour phase from the master-cycle counter.
2377 /// NES NTSC steps the colour phase through 3 frame states (the source of the
2378 /// dot-crawl); PAL/Dendy have no equivalent 3-phase crawl, so the frame
2379 /// parity is exposed instead. Called at each frame boundary.
2380 const fn snapshot_ntsc_phase(&mut self) {
2381 self.frame_ntsc_phase = if matches!(self.region, PpuRegion::Ntsc) {
2382 (self.dot_counter % 3) as u8
2383 } else {
2384 (self.frame & 1) as u8
2385 };
2386 }
2387
2388 /// Current dot (0..=340).
2389 #[must_use]
2390 pub const fn dot(&self) -> u16 {
2391 self.dot
2392 }
2393
2394 /// Current scanline.
2395 #[must_use]
2396 pub const fn scanline(&self) -> i16 {
2397 self.scanline
2398 }
2399
2400 /// Current frame counter.
2401 #[must_use]
2402 pub const fn frame(&self) -> u64 {
2403 self.frame
2404 }
2405
2406 /// Snapshot of CPU-visible register bytes (for the debugger UI).
2407 ///
2408 /// Returns `[ctrl, mask, status, oam_addr]`. Read-only — does NOT clear
2409 /// VBL or toggle the write latch (unlike `cpu_read_register`).
2410 #[must_use]
2411 pub const fn debug_registers(&self) -> [u8; 4] {
2412 [
2413 self.ctrl.bits(),
2414 self.mask.bits(),
2415 self.status.bits(),
2416 self.oam_addr,
2417 ]
2418 }
2419
2420 /// Snapshot of loopy scroll registers `(v, t, x, w)`.
2421 #[must_use]
2422 pub const fn debug_scroll(&self) -> (u16, u16, u8, bool) {
2423 (self.v, self.t, self.x, self.w)
2424 }
2425
2426 /// v1.8.9 — the frame's background scroll `(x, y)` in NES pixels, decoded
2427 /// from the `t` (temp VRAM addr) register + fine-X, including the nametable
2428 /// bits (Mesen HD-pack `_scrollX`/`scrollY`). Used by the HD compositor to
2429 /// offset parallax `<background>` layers by `scroll * ratio`. A frame-level
2430 /// value (the scroll at `t`), not per-scanline. Output-only.
2431 #[must_use]
2432 pub const fn hd_bg_scroll(&self) -> (i32, i32) {
2433 let t = self.t;
2434 let x = ((t & 0x1F) << 3) | (self.x as u16) | if t & 0x0400 != 0 { 0x100 } else { 0 };
2435 let y =
2436 (((t & 0x03E0) >> 2) | ((t & 0x7000) >> 12)) + if t & 0x0800 != 0 { 240 } else { 0 };
2437 (x as i32, y as i32)
2438 }
2439
2440 /// Borrow the 32-byte palette RAM (read-only).
2441 #[must_use]
2442 pub const fn palette_ram(&self) -> &[u8; 32] {
2443 &self.palette_ram
2444 }
2445
2446 /// Borrow OAM (256 bytes = 64 sprites x 4 bytes).
2447 #[must_use]
2448 pub fn oam(&self) -> &[u8] {
2449 &self.oam
2450 }
2451
2452 /// Borrow nametable CIRAM (2 KiB).
2453 #[must_use]
2454 pub fn ciram(&self) -> &[u8] {
2455 &self.ciram
2456 }
2457
2458 /// v2.3.2 "Lucid" — arm or disarm per-byte write attribution.
2459 ///
2460 /// Arming allocates [`crate::provenance::WriteAttribution::HEAP_BYTES`] and
2461 /// starts stamping every subsequent CIRAM / OAM / palette write with the
2462 /// writing instruction's PC and cycle. Disarming frees the store outright, so
2463 /// re-arming starts from a clean slate rather than resurrecting stale records
2464 /// from a previous debugging session.
2465 ///
2466 /// Purely observational — nothing in the render or timing path reads it, so
2467 /// output is bit-identical either way.
2468 #[cfg(feature = "debug-hooks")]
2469 pub fn set_write_attribution(&mut self, enabled: bool) {
2470 self.write_attrib = if enabled {
2471 Some(Box::new(crate::provenance::WriteAttribution::new()))
2472 } else {
2473 None
2474 };
2475 }
2476
2477 /// The write-attribution store, or `None` when not armed.
2478 #[cfg(feature = "debug-hooks")]
2479 #[must_use]
2480 pub fn write_attribution(&self) -> Option<&crate::provenance::WriteAttribution> {
2481 self.write_attrib.as_deref()
2482 }
2483
2484 /// Forget every recorded attribution, keeping the store armed.
2485 ///
2486 /// The core calls this on power-cycle and on save-state restore: the restored
2487 /// bytes were not written by any instruction this session ran, and reporting
2488 /// the PCs that happened to write those offsets *before* the restore would be
2489 /// a confidently wrong answer rather than an absent one.
2490 #[cfg(feature = "debug-hooks")]
2491 pub fn clear_write_attribution(&mut self) {
2492 if let Some(attrib) = self.write_attrib.as_mut() {
2493 attrib.clear();
2494 }
2495 }
2496
2497 /// Push down the `(pc, cycle)` of the CPU instruction whose write is about to
2498 /// land, so the store site can stamp it. Called by the bus immediately before
2499 /// a `$2000-$3FFF` register write and before an OAM DMA burst.
2500 ///
2501 /// A no-op when attribution is not armed, and never read by emulation.
2502 #[cfg(feature = "debug-hooks")]
2503 pub const fn set_attrib_context(&mut self, pc: u16, cycle: u64) {
2504 self.attrib_pc = pc;
2505 self.attrib_cycle = cycle;
2506 }
2507
2508 /// v2.3.2 "Lucid" phase 2 — arm or disarm per-pixel provenance capture.
2509 ///
2510 /// Arming allocates
2511 /// [`crate::provenance::PixelProvenanceFrame::HEAP_BYTES`] and starts
2512 /// recording, for every emitted pixel, the layer that won, the exact palette
2513 /// address, and the nametable / attribute / pattern addresses of the tile
2514 /// actually on screen. Disarming frees the frame.
2515 ///
2516 /// Independent of [`Self::set_write_attribution`]: this says *which bytes*
2517 /// produced a pixel, that says *who wrote* those bytes. The panel wants
2518 /// both, but each is useful alone and neither depends on the other.
2519 ///
2520 /// Output-only, so emulation is bit-identical either way.
2521 #[cfg(feature = "debug-hooks")]
2522 pub fn set_pixel_provenance(&mut self, enabled: bool) {
2523 self.prov_frame = if enabled {
2524 Some(Box::new(crate::provenance::PixelProvenanceFrame::new()))
2525 } else {
2526 None
2527 };
2528 self.prov_armed = enabled;
2529 }
2530
2531 /// The current frame's per-pixel provenance, or `None` when not armed.
2532 #[cfg(feature = "debug-hooks")]
2533 #[must_use]
2534 pub fn pixel_provenance(&self) -> Option<&crate::provenance::PixelProvenanceFrame> {
2535 self.prov_frame.as_deref()
2536 }
2537
2538 /// Forget every recorded pixel, keeping the frame armed.
2539 ///
2540 /// Mirrors [`Self::clear_write_attribution`], and for the same reason: a
2541 /// restore lands mid-frame, so without this the panel would report tile and
2542 /// palette addresses from the abandoned timeline for every pixel above the
2543 /// current scanline, with nothing marking them stale.
2544 #[cfg(feature = "debug-hooks")]
2545 pub fn clear_pixel_provenance(&mut self) {
2546 if let Some(frame) = self.prov_frame.as_mut() {
2547 frame.clear();
2548 }
2549 }
2550
2551 /// Move both provenance stores out, leaving the PPU unarmed.
2552 ///
2553 /// Paired with [`Self::put_provenance`] to carry the stores across a
2554 /// same-timeline restore that would otherwise clear them — see
2555 /// [`crate::provenance::ProvenanceStash`] for why run-ahead needs that and
2556 /// save-state loads and netplay rollback do not.
2557 ///
2558 /// `prov_armed` is dropped to `false` alongside the frame it mirrors, so the
2559 /// invariant "`prov_armed` iff `prov_frame.is_some()`" holds while stashed
2560 /// and `emit_pixel` records nothing into the vacated slot.
2561 #[cfg(feature = "debug-hooks")]
2562 pub const fn take_provenance(&mut self) -> crate::provenance::ProvenanceStash {
2563 let stash = crate::provenance::ProvenanceStash {
2564 write_attrib: self.write_attrib.take(),
2565 prov_frame: self.prov_frame.take(),
2566 prov_armed: self.prov_armed,
2567 };
2568 self.prov_armed = false;
2569 stash
2570 }
2571
2572 /// Put back stores taken by [`Self::take_provenance`].
2573 ///
2574 /// Overwrites whatever is currently held, which is what the pairing wants:
2575 /// the only thing that can have appeared in between is a restore's cleared
2576 /// (or absent) store, and the stashed records are the ones the caller means
2577 /// to keep.
2578 #[cfg(feature = "debug-hooks")]
2579 pub fn put_provenance(&mut self, stash: crate::provenance::ProvenanceStash) {
2580 self.write_attrib = stash.write_attrib;
2581 self.prov_frame = stash.prov_frame;
2582 self.prov_armed = stash.prov_armed;
2583 }
2584
2585 /// Freeze the current instruction context as the cause of an OAM DMA burst.
2586 ///
2587 /// Called by the bus from the `$4014` write, i.e. while
2588 /// [`Self::set_attrib_context`] still holds the `STA $4014` itself.
2589 ///
2590 /// The burst cannot use the live context: `$4014` only arms the transfer, and
2591 /// its 513 or 514 cycles are then stolen from the instructions that follow,
2592 /// so by the time the first OAM byte lands the live context names whichever
2593 /// instruction is being halted — true about the timing, wrong about the cause.
2594 #[cfg(feature = "debug-hooks")]
2595 pub const fn latch_dma_attrib_context(&mut self) {
2596 self.dma_attrib_pc = self.attrib_pc;
2597 self.dma_attrib_cycle = self.attrib_cycle;
2598 }
2599
2600 /// v1.7.0 "Forge" Workstream A1 — debugger writeback: store one palette-RAM
2601 /// byte directly (`idx` masked to 0..32, value masked to the 6-bit palette
2602 /// width), reusing the same canonical mirroring/masking as the live
2603 /// `$2007` write path. Used only by the `debug-hooks` editor writeback,
2604 /// which routes through the gated post-frame poke path — so the default
2605 /// (no-edit) build never calls it and stays byte-identical.
2606 #[cfg(feature = "debug-hooks")]
2607 pub const fn debug_poke_palette(&mut self, idx: u8, value: u8) {
2608 // `palette_index` mirrors $3F10/$14/$18/$1C → $3F00/.. and folds the
2609 // 32-byte window; feed it the raw address so an editor index maps the
2610 // same way a $2007 write would.
2611 let addr = 0x3F00u16 | ((idx & 0x1F) as u16);
2612 let i = palette_index(addr);
2613 self.palette_ram[i] = value & 0x3F;
2614 }
2615
2616 /// v1.7.0 "Forge" Workstream A1 — debugger writeback: store one OAM byte
2617 /// directly. `debug-hooks`-gated; only reached through the gated post-frame
2618 /// poke path, so the default build is byte-identical.
2619 #[cfg(feature = "debug-hooks")]
2620 pub const fn debug_poke_oam(&mut self, idx: u8, value: u8) {
2621 self.oam[idx as usize] = value;
2622 }
2623
2624 /// v1.7.0 "Forge" Workstream A1 — debugger writeback: store one CIRAM byte
2625 /// at a physical offset (caller resolves mirroring via the mapper).
2626 /// `debug-hooks`-gated; only reached through the gated post-frame poke path.
2627 #[cfg(feature = "debug-hooks")]
2628 pub const fn debug_poke_ciram(&mut self, phys: usize, value: u8) {
2629 self.ciram[phys & 0x07FF] = value;
2630 }
2631
2632 /// `true` when sprites are rendered in 8x16 mode (CTRL bit 5).
2633 #[must_use]
2634 pub const fn sprite_size_16(&self) -> bool {
2635 self.ctrl
2636 .contains(crate::registers::PpuCtrl::SPRITE_SIZE_16)
2637 }
2638
2639 /// Base address of the BG pattern table (`$0000` or `$1000`).
2640 #[must_use]
2641 pub const fn bg_pattern_base(&self) -> u16 {
2642 if self
2643 .ctrl
2644 .contains(crate::registers::PpuCtrl::BG_PATTERN_HIGH)
2645 {
2646 0x1000
2647 } else {
2648 0x0000
2649 }
2650 }
2651
2652 /// Base address of the sprite pattern table (8x8 mode only).
2653 #[must_use]
2654 pub const fn sprite_pattern_base(&self) -> u16 {
2655 if self
2656 .ctrl
2657 .contains(crate::registers::PpuCtrl::SPRITE_PATTERN_HIGH)
2658 {
2659 0x1000
2660 } else {
2661 0x0000
2662 }
2663 }
2664
2665 /// OAM DMA byte write: place `value` at `oam[oam_addr]` and increment
2666 /// `oam_addr`. Used by the bus's OAM DMA state machine.
2667 ///
2668 /// Bypasses the OAMADDR-during-rendering corruption modeled by
2669 /// `cpu_write_register` for `$2004` direct writes — DMA writes always
2670 /// hit OAM directly per nesdev.
2671 pub fn oam_dma_write(&mut self, value: u8) {
2672 // v2.9.5 (sibling ledger 3.1c): a DMA byte is a `$2004` write, and
2673 // "writing any value to any PPU port ... will fill this latch"
2674 // (`nesdev_wiki/PPU_registers`). The MiSTer core did this all along.
2675 self.touch_open_bus(value);
2676 self.oam[self.oam_addr as usize] = value;
2677 // v2.3.2 "Lucid" — every byte of the burst is attributed to the ONE
2678 // `STA $4014` that triggered it (the LATCHED context, not the live one:
2679 // the burst steals cycles from the instructions AFTER the trigger, so
2680 // the live context names the halted instruction rather than the cause).
2681 // 256 bytes genuinely share one cause, and reporting anything else would
2682 // invent a history the program does not have.
2683 #[cfg(feature = "debug-hooks")]
2684 if let Some(attrib) = self.write_attrib.as_mut() {
2685 attrib.record_oam(
2686 self.oam_addr,
2687 self.dma_attrib_pc,
2688 self.dma_attrib_cycle,
2689 value,
2690 );
2691 }
2692 // v2.1.4 F2.3 — OAM-decay write hook (no-op at the default): the DMA byte
2693 // recharges the written row's DRAM cells, so refresh its timestamp. Mesen2
2694 // routes DMA writes through the same `WriteSpriteRam` refresh.
2695 self.oam_decay_on_write(self.oam_addr);
2696 self.oam_addr = self.oam_addr.wrapping_add(1);
2697 }
2698
2699 /// Notify the PPU that one CPU cycle has elapsed. Used to drive the
2700 /// post-reset masking window and the open-bus decay timers.
2701 pub const fn on_cpu_cycle(&mut self) {
2702 self.post_reset_mask_remaining = self.post_reset_mask_remaining.saturating_sub(1);
2703 // Open-bus decay: per-bit-group, three independent timers. When a
2704 // group's timer hits 0 those bits clear in the latch. Per
2705 // docs/ppu-2c02.md, real hardware decays in 3-30 ms; we use
2706 // **558.7 ms** — one million CPU cycles at NTSC.
2707 //
2708 // The figure in this comment used to read "~600 ms (≈ 1,073,447 CPU
2709 // cycles at NTSC, rounded to one million)". The arithmetic in the
2710 // parenthesis is right and the headline is not: one million cycles is
2711 // 558.7 ms, so "rounded" was a 7% cut, not a rounding. Corrected here
2712 // because the MiSTer co-simulation DUT has to reproduce this number
2713 // exactly and was quoting the wrong one back.
2714 //
2715 // SWEPT (v2.6.3), because "conservative but well within the window the
2716 // `ppu_open_bus` test cares about" turned out to understate how much
2717 // slack there is. That ROM's decay checks are 100 × `delay_msec 10`
2718 // loops asserting the value has reached zero, so its only real
2719 // constraint is < 1000 ms — and it passes at **3, 10, 20, 30 and
2720 // 100 ms** as well. AccuracyCoin holds 141/141 (RAM decoder), nestest
2721 // matches its golden log, and all eight `nes_blargg` tests pass at
2722 // 30 ms, the documented upper bound.
2723 //
2724 // So this constant is NOT forced by the corpus: a documentation-derived
2725 // value is measurably available. It is kept at one million because
2726 // changing it changes shipped emulator behaviour for every game that
2727 // reads open bus after a long gap, and no test in this repository can
2728 // adjudicate which is right — the wiki's band and this model differ by
2729 // ~19×, and neither has an independent oracle here. Recorded as a
2730 // deliberate hold rather than a derivation. See
2731 // `RustyNES_MiSTer/docs/rung3-ppu.md`, where the same constant is
2732 // reproduced as 3,000,000 dots and the DUT-side sweep is written up.
2733 // NOTE (v2.3.1 G5): reformulating this as a deadline comparison instead
2734 // of a per-cycle decrement was measured by DELETING the loop outright —
2735 // the ceiling any reformulation could reach — and the ceiling is ZERO.
2736 // ~29,780 calls/frame sounds expensive; it is three predictable
2737 // compare-and-decrement steps on data already in L1, which an
2738 // out-of-order core absorbs entirely. Do not re-attempt; see
2739 // `docs/performance.md`.
2740 let mut i = 0;
2741 while i < 3 {
2742 if self.open_bus_decay[i] > 0 {
2743 self.open_bus_decay[i] -= 1;
2744 if self.open_bus_decay[i] == 0 {
2745 self.open_bus &= !Self::OPEN_BUS_GROUP_MASKS[i];
2746 }
2747 }
2748 i += 1;
2749 }
2750 }
2751
2752 /// Per-bit-group masks for the open-bus latch decay model. Group 0 is
2753 /// bits 0-4 (refreshed by writes, $2004 reads, and $2007 reads — both
2754 /// palette and non-palette). Group 1 is bit 5 (refreshed by writes,
2755 /// $2002 reads, $2004 reads, and $2007 reads). Group 2 is bits 6-7
2756 /// (refreshed by writes, $2002 reads, $2004 reads, and $2007 non-palette
2757 /// reads — but not by palette reads).
2758 const OPEN_BUS_GROUP_MASKS: [u8; 3] = [0x1F, 0x20, 0xC0];
2759
2760 /// Decay-timer reload value (~600 ms at NTSC).
2761 const OPEN_BUS_DECAY_RELOAD: u32 = 1_000_000;
2762
2763 /// Refresh the open-bus latch. `group_mask` is a bitmap selecting which
2764 /// of the three decay groups to refresh: bit 0 = bits 0-4, bit 1 = bit 5,
2765 /// bit 2 = bits 6-7. Only the bits in those groups are copied from
2766 /// `value`; bits in groups not selected retain their previous latch value
2767 /// and their decay timer is left to keep counting down.
2768 const fn refresh_open_bus(&mut self, value: u8, group_mask: u8) {
2769 let mut i = 0;
2770 while i < 3 {
2771 if (group_mask >> i) & 1 == 1 {
2772 let m = Self::OPEN_BUS_GROUP_MASKS[i];
2773 self.open_bus = (self.open_bus & !m) | (value & m);
2774 self.open_bus_decay[i] = Self::OPEN_BUS_DECAY_RELOAD;
2775 }
2776 i += 1;
2777 }
2778 }
2779
2780 /// Refresh **all** bit groups of the open-bus latch — used by writes and
2781 /// any read that drives all 8 bits (e.g. $2004 OAMDATA, $2007 non-palette
2782 /// PPUDATA).
2783 const fn touch_open_bus(&mut self, value: u8) {
2784 self.refresh_open_bus(value, 0b111);
2785 }
2786
2787 /// Number of PPU dots between a `$2002` read's start (M2 high, when the
2788 /// VBL flag is latched) and its end (M2 low, when the *unlatched*
2789 /// sprite-0-hit / overflow flags are sampled). On a revision-G 2A03 M2's
2790 /// 15/24 duty cycle puts read-end ~1.875 PPU dots after read-start; we
2791 /// round to the nearest whole dot for the lockstep computed sample. This
2792 /// is the empirically-tuned knob for the `$2002 flag timing` test.
2793 ///
2794 /// Tuned to **1**: with the test's reads spaced 1 PPU dot apart, a 1-dot
2795 /// window yields the primary answer key `$E0,$E0,$80,$00` (read 3 = `$80`:
2796 /// VBL latched set, sprite flags read 0). A 2-dot window also passes Test 1
2797 /// (via the alt key `$E0,$80,$80,$00`) but masks one extra read position,
2798 /// which regresses `$2004 Stress` (whose `$2002` sync read lands there).
2799 const STATUS_READ_END_DOTS: u16 = 1;
2800
2801 /// True when advancing `dots` PPU dots from the current `(scanline, dot)`
2802 /// reaches or passes the pre-render dot-1 flag-clear, where the
2803 /// sprite-0-hit and overflow flags are cleared. Used by the `$2002`
2804 /// two-point read model to sample bits 6/5 as-of read-end. Forward
2805 /// distance only: a position already at/after this frame's clear is *not*
2806 /// "imminent" (its flags are already cleared in the status register, and
2807 /// the full-frame wrap distance never satisfies the small `dots` bound).
2808 fn sprite_flags_clear_imminent(&self, dots: u16) -> bool {
2809 const DOTS_PER_LINE: i32 = 341;
2810 let lines = i32::from(self.region.prerender_line()) + 1;
2811 let frame_dots = lines * DOTS_PER_LINE;
2812 let target = i32::from(self.region.prerender_line()) * DOTS_PER_LINE + 1;
2813 let cur = i32::from(self.scanline) * DOTS_PER_LINE + i32::from(self.dot);
2814 let until = (target - cur).rem_euclid(frame_dots);
2815 until > 0 && until <= i32::from(dots)
2816 }
2817
2818 /// CPU register read at `$2000-$3FFF` (only the low 3 bits matter).
2819 #[allow(clippy::too_many_lines)]
2820 pub fn cpu_read_register<B: PpuBus>(&mut self, reg: u8, bus: &mut B) -> u8 {
2821 // 2C05 swaps $2000<->$2001 (no-op on every other PPU).
2822 match self.map_register(reg) {
2823 // $2000 / $2001 / $2003 / $2005 / $2006 are write-only; reads
2824 // return open-bus.
2825 0 | 1 | 3 | 5 | 6 => self.open_bus,
2826 2 => {
2827 // $2002 PPUSTATUS. High 3 bits are real; low 5 are open-bus.
2828 // `v` is the value as-of read-start (M2 high); it both feeds the
2829 // open-bus I/O-latch refresh below and is the base for the
2830 // CPU-visible return value. The Tier-1.1 read-end sample (below)
2831 // is applied ONLY to the returned byte, NOT to the open-bus
2832 // refresh — keeping the decay model byte-identical so real games
2833 // that read open bus after a `$2002` read are unaffected.
2834 // On a 2C05 the low 5 bits return the PPU identifier (copy
2835 // protection) instead of PPU open bus; on every other PPU they
2836 // are the open-bus latch (byte-identical to the legacy path).
2837 let low5 = if self.is_2c05 {
2838 self.id_2c05 & 0x1F
2839 } else {
2840 self.open_bus & 0x1F
2841 };
2842 let v = (self.status.bits() & 0xE0) | low5;
2843 // Clear VBL and the w toggle as a side effect.
2844 self.status.remove(PpuStatus::VBLANK);
2845 self.w = false;
2846 // R2 (master-clock R1 substrate): a $2002 read drops the /NMI
2847 // line UNCONDITIONALLY (Mesen2 `UpdateStatusFlag:588`
2848 // `ClearNmiFlag()`; TetaNES `read_status: nmi_pending=false`).
2849 // Correct under R1's on-time access (the read lands at the
2850 // access's exact dot); the CPU's φ2 edge detector sees the
2851 // level fall on the same access.
2852 {
2853 self.nmi_line = false;
2854 }
2855 // Race: reading PPUSTATUS at exactly the cycle VBL would
2856 // have been set suppresses VBL + NMI for that frame. We
2857 // approximate the race window as scanline 241 dot 0 (the
2858 // dot before set) and dot 1 (the set dot).
2859 //
2860 // Session-18 / C1 attempt 16 (PPU-axis) investigated
2861 // tightening the predicate from `dot <= 1` to `dot == 0`
2862 // (matching Mesen2's `Core/NES/NesPpu.cpp::
2863 // UpdateStatusFlag()` `_cycle == 0` strict 1-dot window
2864 // and the nesdev wiki spec) but ROLLED IT BACK because
2865 // it did NOT flip the failing `cpu_interrupts_v2/{2,3,5}`
2866 // tests. The empirical oracle in
2867 // `ppu::tests::vbl_race_window_2002_read_sweep` (added
2868 // in Session-18) shows the predicate change cleanly
2869 // narrows the window AT the unit-test layer, but the
2870 // load-bearing axis at the integration-test layer is the
2871 // CPU-vs-PPU per-cycle access interleaving — a deeper
2872 // architectural surface that the Session-13 cold-boot
2873 // alignment closed at the FRAME-anchor level but NOT at
2874 // the INTRA-CYCLE phase level. See
2875 // `docs/audit/session-18-c1-attempt16-ppu-axis-rollback-2026-05-22.md`
2876 // and ADR-0002 §"Decision update (2026-05-22, Session-18)".
2877 // C1 attempt 18 (coordinated with the CPU-side φ1/φ2
2878 // split): when the access-reorder feature is enabled,
2879 // narrow the suppression window from `dot <= 1` to
2880 // `dot == 0` per Mesen2 line 590 + nesdev wiki spec.
2881 // The CPU-side shift puts our BIT $2002 reads at dot 1
2882 // (post-φ1-tick) instead of dot 0, and the dot-1 read
2883 // should NOT trigger suppression (it sees the
2884 // just-set VBL). Both changes together close the
2885 // `cpu_interrupts_v2/{2,3,5}` sync_vbl divergence
2886 // documented in Session-17/18 audits.
2887 // R2 (mc-r1-substrate) narrows the race window to `dot == 0`
2888 // (Mesen2 `:590`) — on the on-time substrate a dot-1 read is a
2889 // normal post-set read; only dot-0 (one PPU clock before VBL
2890 // set) arms suppression.
2891 let in_race_window =
2892 self.scanline == self.region.vblank_start_line() && self.dot == 0;
2893 if in_race_window {
2894 self.suppress_vbl_this_frame = true;
2895 // If NMI was already raised on dot 1 this same cycle,
2896 // pull it back down too.
2897 self.nmi_line = false;
2898 }
2899 // Reading PPUSTATUS only refreshes the upper 3 bits of the
2900 // open-bus latch (the bits sourced from the status register);
2901 // the lower 5 bits retain both their previous value AND their
2902 // decay timer. See nesdev wiki "PPU registers" §"Open bus",
2903 // `cpu_dummy_writes_ppumem` test ROM (open_bus_read_test 2),
2904 // and `ppu_open_bus.nes` test 7
2905 // ("Reading $2002 shouldn't refresh low 5 bits of decay value").
2906 // Refresh groups 1 (bit 5) and 2 (bits 6-7) only.
2907 // Uses the read-start value `v` (the Tier-1.1 read-end mask is
2908 // applied to the return value only, below).
2909 self.refresh_open_bus(v, 0b110);
2910 // v2.0 Tier 1.1 — $2002 two-point intra-read flag sampling.
2911 // VBL (bit 7) is latched at read-start (M2 high) = the current
2912 // dot, already captured in `v`. The sprite-0 (bit 6) and
2913 // overflow (bit 5) flags are NOT latched; the CPU samples them
2914 // at read-end (M2 low), ~1.875 PPU dots later. A read straddling
2915 // the pre-render dot-1 flag-clear therefore returns VBL still set
2916 // while the sprite flags already read 0 — the AccuracyCoin
2917 // `$2002 flag timing` answer key `$E0,$E0,$80,$00`. Mask bits 6/5
2918 // on the returned byte when read-end lands at/after the clear
2919 // (TriCNES `EmulateUntilEndOfRead`). Returned-value-only so the
2920 // open-bus latch stays byte-identical for real games.
2921 if self.sprite_flags_clear_imminent(Self::STATUS_READ_END_DOTS) {
2922 return v & !0x60;
2923 }
2924 v
2925 }
2926 4 => {
2927 // $2004 OAMDATA. Returns OAM[OAMADDR] without auto-increment.
2928 // Sprite attribute bytes (every 4th byte starting at offset 2)
2929 // have bits 2-4 unimplemented in OAM and always read as 0,
2930 // even though writes can store them. See nesdev wiki "PPU
2931 // OAM" → "Byte 2 (attributes)".
2932 //
2933 // v2.0 Tier 1.2: while the screen is being drawn on a visible
2934 // scanline, $2004 returns the value the PPU is currently using
2935 // for sprite evaluation / loading (the OAM data bus), NOT
2936 // OAM[OAMADDR]. The isolated `ppu-oam-data-bus` model
2937 // (`oam_data_bus_read`) reproduces this per AccuracyCoin
2938 // `$2004 Stress`; see Mesen2 `NesPpu.cpp:298-313/361-380`.
2939 {
2940 if self.oam_data_bus_is_live() {
2941 let v = self.oam_data_bus_read();
2942 self.touch_open_bus(v);
2943 return v;
2944 }
2945 }
2946 // v2.1.4 F2.3 — OAM-decay read hook (no-op at the default): a
2947 // non-rendering `$2004` read refreshes the row, or returns the
2948 // decayed pattern if it has gone stale. Must run before the read.
2949 self.oam_decay_on_read(self.oam_addr);
2950 let mut v = self.oam[self.oam_addr as usize];
2951 if (self.oam_addr & 0x03) == 0x02 {
2952 v &= 0xE3;
2953 }
2954 // Per nesdev wiki "PPU registers" §$2004 + AccuracyCoin
2955 // "Address $2004 behavior" sub-tests 4 + 9: during dots
2956 // 1-64 of every rendered scanline (the secondary-OAM
2957 // clear phase) AND during dots 257-320 (the sprite-tile-
2958 // loading interval — also when the secondary-OAM bytes
2959 // are being read out to the shift registers), $2004
2960 // reads return $FF.
2961 //
2962 // (When `ppu-oam-data-bus` is on, the rendering case returns
2963 // above; this fallback covers the flag-off build + the
2964 // non-rendering paths.)
2965 if self.is_render_scanline()
2966 && self.mask.rendering_enabled()
2967 && ((1..=64).contains(&self.dot) || (257..=320).contains(&self.dot))
2968 {
2969 v = 0xFF;
2970 }
2971 self.touch_open_bus(v);
2972 v
2973 }
2974 7 => {
2975 // Diagnostic: log where each $2007 read lands (scanline/dot/mask).
2976 if (4400..6200).contains(&self.frame) {
2977 use core::sync::atomic::Ordering::Relaxed;
2978 let i = read2007_diag::IDX.fetch_add(1, Relaxed) as usize;
2979 if i < 1024 {
2980 #[allow(clippy::cast_sign_loss)]
2981 let sl = ((i32::from(self.scanline) + 1) as u32) & 0x1FF;
2982 let packed = (sl << 18)
2983 | ((u32::from(self.dot) & 0x1FFF) << 5)
2984 | (u32::from(self.mask.rendering_enabled()) << 1)
2985 | u32::from(self.is_render_scanline());
2986 read2007_diag::LOG[i].store(packed, Relaxed);
2987 }
2988 }
2989 // $2007 PPUDATA. Buffered for $0000-$3EFF; palette reads
2990 // bypass the buffer but still update it with the underlying
2991 // nametable mirror.
2992 let addr = self.v & 0x3FFF;
2993 let is_palette = addr >= 0x3F00;
2994 // W2: set when THIS read arms the PPUDATA SM countdown with the
2995 // defer-v-inc sub-knob on — the v-glitch increment then happens
2996 // at the TStep (the countdown landing dot), not here.
2997 let mut defer_v_inc = false;
2998 let result = if is_palette {
2999 // Palette read: high 2 bits = open bus.
3000 let palette = self.read_palette(addr);
3001 let v_with_open_bus = (palette & 0x3F) | (self.open_bus & 0xC0);
3002 // Buffer gets the underlying nametable byte (from CIRAM).
3003 self.data_buffer = self.read_vram(bus, addr & 0x2FFF);
3004 v_with_open_bus
3005 } else {
3006 let r = self.data_buffer;
3007 // During rendering the buffer is NOT loaded from a read at
3008 // `v`. TriCNES (`Emulator.cs` `PPU_DATA_StateMachine` +
3009 // the `$2007` CPU read): the read-END arms a latch cascade
3010 // and the actual `PPU_ReadBuffer` reload happens ~4 dots
3011 // later, latching the value the BG/sprite FETCH cadence
3012 // drove on the VRAM bus at the LANDING dot (the fetch has
3013 // bus priority). Modeled as a PPU-dot countdown consumed
3014 // in `Ppu::tick`; the returned value stays the OLD buffer
3015 // (the priming-read contract). Delay 0 = immediate latch
3016 // of the current bus value.
3017 if self.mask.rendering_enabled() && self.is_render_scanline() {
3018 octal_trace::push(
3019 octal_trace::K_R2007,
3020 self.frame,
3021 self.scanline,
3022 self.dot,
3023 u32::from(self.v & 0x3FFF),
3024 );
3025 let n = read2007_diag::RENDER_BUFFER_DOT_DELAY
3026 .load(core::sync::atomic::Ordering::Relaxed)
3027 as u8;
3028 if n == 0 {
3029 self.data_buffer = self.render_data_bus;
3030 } else {
3031 self.ppudata_sm_countdown = n;
3032 // TriCNES `PPU_DATA_StateMachine_Half`: the TStep
3033 // (v-glitch increment) fires at the SAME dot as the
3034 // PD_RB buffer reload, AFTER it — so the fetches in
3035 // the read-to-reload window still use the OLD `v`.
3036 if read2007_diag::RENDER_BUFFER_DEFER_V_INC
3037 .load(core::sync::atomic::Ordering::Relaxed)
3038 != 0
3039 {
3040 self.ppudata_v_inc_pending = true;
3041 defer_v_inc = true;
3042 }
3043 }
3044 } else {
3045 self.data_buffer = self.read_vram(bus, addr);
3046 }
3047 r
3048 };
3049 // Per nesdev "PPU rendering"
3050 // (https://www.nesdev.org/wiki/PPU_scrolling#$2007_reads_and_writes_during_rendering):
3051 // "Reading or writing PPUDATA during rendering (on the
3052 // pre-render line and the visible lines 0-239, only when
3053 // rendering is enabled) does not increment the address
3054 // normally, but instead increments both coarse X scroll
3055 // and Y scroll simultaneously, with normal wrapping."
3056 // This is the canonical "$2007 read w/ rendering" quirk
3057 // that AccuracyCoin's `PPU Behavior :: $2007 read w/
3058 // rendering` Test 2 brackets.
3059 //
3060 // W2 (`mc-ppu-2007-render-buffer` + defer-v-inc sub-knob): when
3061 // this read armed the PPUDATA SM countdown, the increment is
3062 // performed at the TStep (the countdown landing dot in
3063 // `Ppu::tick`) instead of here.
3064 let apply_inc_now = !defer_v_inc;
3065 if apply_inc_now {
3066 if self.mask.rendering_enabled() && self.is_render_scanline() {
3067 self.inc_hori_v();
3068 self.inc_vert_v();
3069 } else {
3070 let inc = if self.ctrl.contains(PpuCtrl::VRAM_INCREMENT_32) {
3071 32
3072 } else {
3073 1
3074 };
3075 self.v = self.v.wrapping_add(inc) & 0x7FFF;
3076 }
3077 }
3078 // A12 transition can occur here.
3079 self.observe_a12(bus);
3080 if is_palette {
3081 // Palette reads only refresh bits 0-5 of the decay model
3082 // (palette is 6-bit); bits 6-7 retain their previous
3083 // value AND timer. Required by `ppu_open_bus.nes` test 9.
3084 self.refresh_open_bus(result, 0b011);
3085 } else {
3086 self.touch_open_bus(result);
3087 }
3088 result
3089 }
3090 _ => unreachable!(),
3091 }
3092 }
3093
3094 /// CPU register write.
3095 // Large by nature: an 8-way `$2000-$2007` register-dispatch match, each arm
3096 // carrying its own hardware-quirk handling (the v2.0.2 octal-latch `$2006`
3097 // hook nudged it past the 100-line lint threshold).
3098 #[allow(clippy::too_many_lines)]
3099 pub fn cpu_write_register<B: PpuBus>(&mut self, reg: u8, value: u8, bus: &mut B) {
3100 // Open-bus latch always picks up the written value.
3101 self.touch_open_bus(value);
3102 // 2C05 swaps $2000<->$2001 (no-op on every other PPU).
3103 match self.map_register(reg) {
3104 0 => {
3105 // $2000 PPUCTRL.
3106 if self.post_reset_mask_remaining > 0 {
3107 return;
3108 }
3109 let prev_nmi_enable = self.ctrl.contains(PpuCtrl::NMI_ENABLE);
3110 self.ctrl = PpuCtrl::from_bits_truncate(value);
3111 // t bits 11-10 = nametable bits 1-0.
3112 self.t = (self.t & 0xF3FF) | ((u16::from(value) & 0x03) << 10);
3113 // NMI bit 0->1 transition while VBL set asserts NMI immediately.
3114 let new_nmi_enable = self.ctrl.contains(PpuCtrl::NMI_ENABLE);
3115 if !prev_nmi_enable && new_nmi_enable && self.status.contains(PpuStatus::VBLANK) {
3116 self.nmi_line = true;
3117 }
3118 if !new_nmi_enable {
3119 // Disabling NMI lowers the line.
3120 self.nmi_line = false;
3121 }
3122 }
3123 1 => {
3124 // $2001 PPUMASK.
3125 if self.post_reset_mask_remaining > 0 {
3126 return;
3127 }
3128 let was_rendering = self.mask.rendering_enabled();
3129 self.mask = PpuMask::from_bits_truncate(value);
3130 self.arm_oam_corruption_disable(was_rendering);
3131 // v2.0 Phase 6 (mc-ppu-subpos): arm the analog `$2001` BG-reload
3132 // delay. `self.mask` (and so the sprite-eval / shift / pixel
3133 // path) updates IMMEDIATELY — only the BG shift-register RELOAD
3134 // is gated on a value delayed `MASK_WRITE_DELAY` dots behind the
3135 // mask (TriCNES gates `PPU_Render_ShiftRegistersAndBitPlanes` ->
3136 // the reload on `PPU_Mask_Show*_Delayed`, while the per-half-dot
3137 // SHIFT runs on the IMMEDIATE mask). On a render re-enable edge
3138 // the shifter therefore advances for several dots (injecting the
3139 // serial-in '1') BEFORE the reload resumes — so one reload is
3140 // SKIPPED and the accumulated '1's reach the output (BG Serial
3141 // In) WITHOUT perturbing the sprite path (Stale Sprite Shift
3142 // Regs) or normal rendering (the reload value latched between
3143 // toggles equals the live mask -> byte-identical).
3144 // Freeze the BG-reload gate at its prior value for the analog
3145 // write-delay window; `tick` re-syncs it to the live mask once
3146 // the countdown settles.
3147 {
3148 self.mask_write_delay =
3149 MASK_WRITE_DELAY.load(core::sync::atomic::Ordering::Relaxed);
3150 }
3151 }
3152 2 => {
3153 // $2002 is read-only; writes only update the open-bus latch
3154 // (already done above) and otherwise have no effect.
3155 }
3156 3 => {
3157 // $2003 OAMADDR.
3158 self.oam_addr = value;
3159 // v2.1.7 P5 — OAMADDR ($2003) write-during-rendering OAM
3160 // corruption, modeled only on the earlier `Rp2c02G` revision
3161 // (default `Rp2c02H` skips this entirely → byte-identical). On
3162 // real "rev E+" 2C02 silicon, writing $2003 while rendering is
3163 // active corrupts one OAM "row"; RustyNES arms the shared
3164 // `CorruptOAM` row-copy (see `process_oam_corruption`) targeting
3165 // the row the write's high bits select, committed on the next
3166 // rendered dot. The `!oam_corruption_pending` guard defers to an
3167 // already-armed corruption (e.g. the rendering-disable model) so
3168 // the two sources never race. See `docs/ppu-2c02.md` (§OAMADDR
3169 // corruption) and `docs/accuracy-ledger.md` for the honesty note.
3170 if self.die_revision.models_oamaddr_corruption()
3171 && self.mask.rendering_enabled()
3172 && self.is_render_scanline()
3173 && !self.oam_corruption_pending
3174 {
3175 self.oam_corruption_pending = true;
3176 self.oam_corruption_index = (value >> 3) & 0x1F;
3177 }
3178 }
3179 4 => {
3180 // $2004 OAMDATA write. Per nesdev §PPU OAM:
3181 //
3182 // - Outside rendering (or rendering disabled): write the
3183 // value to OAM[OAMADDR] and increment OAMADDR by 1.
3184 // - During rendering (visible / pre-render scanline with
3185 // rendering enabled): the write is BLOCKED (real chip
3186 // does a glitchy "OAM read" instead, value discarded),
3187 // but OAMADDR is still incremented by **4** (NOT 1) —
3188 // the silicon's OAMADDR-bump-on-rendering-write quirk
3189 // that AccuracyCoin's `Sprite Evaluation :: Misaligned
3190 // OAM behavior` test (T-60-002, 2026-05-17) brackets.
3191 //
3192 // Pre-fix our impl always incremented by 1; matches the
3193 // outside-rendering path but is wrong during rendering.
3194 if self.mask.rendering_enabled() && self.is_render_scanline() {
3195 // During-rendering quirk: OAMADDR += 4, then mask
3196 // with $FC (clear bottom 2 bits — re-align to a
3197 // 4-byte sprite boundary). Required for
3198 // AccuracyCoin's "Address $2004 behavior" sub-test
3199 // A which writes $2004 with OAMADDR=1 during
3200 // rendering, then expects subsequent reads at
3201 // OAMADDR=4 (= (1+4) & $FC) to read OAM[4].
3202 self.oam_addr = self.oam_addr.wrapping_add(4) & 0xFC;
3203 } else {
3204 self.oam[self.oam_addr as usize] = value;
3205 // v2.3.2 "Lucid" — attribute only the branch that actually
3206 // stores. The during-rendering branch above is BLOCKED by the
3207 // hardware quirk, so recording it would attribute a byte to an
3208 // instruction that demonstrably did not write it.
3209 #[cfg(feature = "debug-hooks")]
3210 if let Some(attrib) = self.write_attrib.as_mut() {
3211 attrib.record_oam(self.oam_addr, self.attrib_pc, self.attrib_cycle, value);
3212 }
3213 // v2.1.4 F2.3 — OAM-decay write hook (no-op at the default):
3214 // a direct `$2004` write refreshes the written row.
3215 self.oam_decay_on_write(self.oam_addr);
3216 self.oam_addr = self.oam_addr.wrapping_add(1);
3217 }
3218 }
3219 5 => {
3220 // $2005 PPUSCROLL.
3221 if self.post_reset_mask_remaining > 0 {
3222 return;
3223 }
3224 if self.w {
3225 // Second write — Y scroll.
3226 self.t = (self.t & 0x8C1F)
3227 | ((u16::from(value) & 0xF8) << 2)
3228 | ((u16::from(value) & 0x07) << 12);
3229 self.w = false;
3230 } else {
3231 // First write — X scroll.
3232 self.t = (self.t & 0xFFE0) | (u16::from(value) >> 3);
3233 self.x = value & 0x07;
3234 self.w = true;
3235 }
3236 }
3237 6 => {
3238 // $2006 PPUADDR.
3239 if self.post_reset_mask_remaining > 0 {
3240 return;
3241 }
3242 if self.w {
3243 // Second write — low byte; copy t to v.
3244 self.t = (self.t & 0xFF00) | u16::from(value);
3245 // v2.0.3 (ADR 0030, Option 1) — "Hybrid Addresses" the natural
3246 // way. During rendering, a `$2006` second write does NOT copy
3247 // `t -> v` immediately; it stages the delayed-`CopyV` countdown
3248 // (`TriCNES` `PPU_Update2006Delay`). The `v = t` and the
3249 // `address_bus = v` splice happen when the countdown lands
3250 // (`Self::tick`), by which point the fetch cadence has advanced
3251 // coarse-X and the per-group phase-0 nametable ALE has NATURALLY
3252 // loaded `octal_latch` with the one-tile-ahead NT-low (`$19`),
3253 // so the landing read splices `$2F00 | $19 = $2F19` with no
3254 // reconstruction. Outside rendering the copy is immediate (the
3255 // delay is unobservable there and would only risk shifting
3256 // tightly-timed non-render code), so non-render behavior is
3257 // unchanged.
3258 let deferred_copy_v = self.mask.rendering_enabled()
3259 && self.is_render_scanline()
3260 // Only within the active BG-fetch window (visible dots
3261 // 1..=256 + the dots-321..=336 prefetch). The "Hybrid
3262 // Addresses" corruption can ONLY manifest when a background
3263 // fetch is in flight to consume the stale octal latch; a
3264 // `$2006` write during the sprite/HBlank interval
3265 // (257..=320) has no BG-fetch consumer, so deferring `v = t`
3266 // there would serve no accuracy purpose and only risk
3267 // shifting the many commercial mid-frame scroll splits
3268 // (SMB3's status-bar `$2006`/`$2005`, MMC5 titles) that
3269 // write during HBlank. Narrowing to the fetch window keeps
3270 // the delayed-`CopyV` surgical to the modeled artifact.
3271 && ((1..=256).contains(&self.dot) || (321..=336).contains(&self.dot))
3272 && {
3273 // Alignment-dependent delay (TriCNES uses 4 for three of
3274 // four CPU/PPU phases, 5 for one). The `$2006` write is
3275 // applied at the START of a CPU cycle in RustyNES's
3276 // lockstep bus (before that cycle's 3 PPU ticks); the
3277 // corrupted NT read is the phase-1 dot of the fetch group
3278 // one coarse-X past the write. Empirically calibrated
3279 // against the TriCNES per-dot trace (see the campaign
3280 // plan); the AccuracyCoin test also retries across
3281 // frames/alignments so at least one alignment lands.
3282 self.copy_v_delay = COPY_V_DELAY;
3283 octal_trace::push(
3284 octal_trace::K_W2006,
3285 self.frame,
3286 self.scanline,
3287 self.dot,
3288 u32::from(self.t & 0x3FFF),
3289 );
3290 true
3291 };
3292 if !deferred_copy_v {
3293 self.v = self.t;
3294 // PPUADDR write can flip A12.
3295 self.observe_a12(bus);
3296 }
3297 self.w = false;
3298 } else {
3299 // First write — high byte (clears bit 14 of t).
3300 self.t = (self.t & 0x00FF) | ((u16::from(value) & 0x3F) << 8);
3301 self.w = true;
3302 }
3303 }
3304 7 => {
3305 // $2007 PPUDATA write. Same rendering quirk as the
3306 // read path (see `cpu_read_register` case 7 docstring):
3307 // writes during rendering increment both coarse-X and
3308 // Y scroll instead of the normal `inc` value.
3309 let addr = self.v & 0x3FFF;
3310 if addr >= 0x3F00 {
3311 self.write_palette(addr, value);
3312 } else {
3313 self.write_vram(bus, addr, value);
3314 }
3315 if self.mask.rendering_enabled() && self.is_render_scanline() {
3316 self.inc_hori_v();
3317 self.inc_vert_v();
3318 } else {
3319 let inc = if self.ctrl.contains(PpuCtrl::VRAM_INCREMENT_32) {
3320 32
3321 } else {
3322 1
3323 };
3324 self.v = self.v.wrapping_add(inc) & 0x7FFF;
3325 }
3326 self.observe_a12(bus);
3327 }
3328 _ => unreachable!(),
3329 }
3330 }
3331
3332 /// Address-bus A12 = `v` bit 12 during `$0000-$3FFF` accesses. Notify the
3333 /// mapper on every transition. Also called by `observe_a12_addr` for
3334 /// the actual pattern fetch addresses (background and sprite fetches
3335 /// directly read CHR via the address bus, not via `v`).
3336 fn observe_a12<B: PpuBus>(&mut self, bus: &mut B) {
3337 let level = (self.v & 0x1000) != 0;
3338 if level != self.last_a12_level {
3339 bus.notify_a12(level);
3340 self.last_a12_level = level;
3341 }
3342 }
3343
3344 /// Report a background fetch group's A12 level where the `NESdev` MMC3
3345 /// page puts it: the pattern fetches' high level at group phase 3 (dot
3346 /// 324 for the next line's first prefetched tile, dot 4 for a line's own
3347 /// first group), one dot before the pattern-low fetch's ALE dot, and the
3348 /// return to the nametable fetch's low level at phase 7.
3349 ///
3350 /// The page: "if the BG uses `$1000`, and the sprites use `$0000`, the
3351 /// IRQ counter should decrement on PPU cycle 324 of the previous
3352 /// scanline", and for the opposite arrangement "on PPU cycle 260". The
3353 /// sprite path already reported at that convention (its rises land at
3354 /// 260, 268, ... 316). The background reported at its READ dots (phase
3355 /// 5 for the pattern, phase 1 for the nametable), two dots later than
3356 /// the page, so a rise that landed on the first dot of a CPU cycle's
3357 /// catch-up was seen by an MMC3 a cycle late. blargg `4-scanline_timing`
3358 /// failed at sub-test 9 ("Scanline 0 IRQ should occur sooner when
3359 /// `$2000=$10`") on exactly that case; with this it fails at 12, the
3360 /// sub-test the `MiSTer` DUT fails (T-MMC3-BG-A12).
3361 ///
3362 /// The read halves still report their own addresses, and with the level
3363 /// already set those reports change nothing. Only the level the mapper
3364 /// sees, and when, moves: with the background at `$0000` (almost every
3365 /// MMC3 game) both reports here are low and nothing changes.
3366 fn observe_bg_a12_lead<B: PpuBus>(&mut self, bus: &mut B, phase: u16) {
3367 match phase {
3368 3 => {
3369 let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
3370 self.observe_a12_addr(bus, bg_table);
3371 }
3372 7 => self.observe_a12_addr(bus, 0x2000),
3373 _ => {}
3374 }
3375 }
3376
3377 /// Notify the mapper of an A12 transition implied by an explicit
3378 /// pattern-table fetch address (BG / sprite fetches that bypass `v`).
3379 fn observe_a12_addr<B: PpuBus>(&mut self, bus: &mut B, addr: u16) {
3380 let level = (addr & 0x1000) != 0;
3381 if level != self.last_a12_level {
3382 bus.notify_a12(level);
3383 self.last_a12_level = level;
3384 }
3385 }
3386
3387 /// Read from PPU memory `$0000-$3EFF` honoring CIRAM ownership: CHR
3388 /// (`$0000-$1FFF`) goes to the bus/mapper; nametable (`$2000-$3EFF`)
3389 /// reads come from the PPU-owned CIRAM through the mapper-supplied
3390 /// mirroring map.
3391 ///
3392 /// The bus is consulted via `peek_nametable` first; mappers like MMC5
3393 /// in fill mode or ExRAM-as-nametable mode synthesize the byte
3394 /// directly. Only when the bus declines (`None`) do we hit CIRAM.
3395 // `&mut self` is required under `mc-ppu-2007-render-buffer` (it latches
3396 // `self.render_data_bus` below); clippy's needless-pass-by-ref-mut only fires
3397 // on the default build where that cfg is off, so allow it here.
3398 fn read_vram<B: PpuBus>(&mut self, bus: &mut B, addr: u16) -> u8 {
3399 // v2.9.7 — every read drives its address onto the PPU bus, so every
3400 // read reports its A12 level. Before v2.9.7 only pattern fetches and
3401 // `v` did, so the garbage nametable reads of the sprite window never
3402 // pulled A12 low, and the eight A12 pulses of each rendered line
3403 // reached the mapper as one. MMC3's filter ignores the short lows, so
3404 // it never showed there. Boards that count raw edges were starved
3405 // eightfold: Acclaim's MC-ACC (a divide-by-8 on every edge) and
3406 // mapper 91 submapper 0 ("64 unfiltered rises"). Pinned by
3407 // `a12_reports_the_hardware_stream_and_an_mmc3_filter_still_sees_241`.
3408 // `observe_a12_addr` calls the bus only on a change of level.
3409 self.observe_a12_addr(bus, addr & 0x3FFF);
3410 self.read_vram_unobserved(bus, addr)
3411 }
3412
3413 /// [`Self::read_vram`] without the A12 report: the read itself, for the
3414 /// one caller whose A12 edge would land in the wrong order (the sprite
3415 /// window's second garbage nametable read; see `tick_sprite_fetch_read`).
3416 #[allow(clippy::needless_pass_by_ref_mut)]
3417 fn read_vram_unobserved<B: PpuBus>(&mut self, bus: &mut B, addr: u16) -> u8 {
3418 let a = addr & 0x3FFF;
3419 // Every PPU bus read passes through here -- pattern fetches, nametable
3420 // and attribute fetches, sprite pattern fetches, and `$2007`. Capturing
3421 // at the choke point rather than at each call site is what makes the
3422 // trace complete by construction: a fetch added later cannot forget to
3423 // record itself.
3424 #[cfg(feature = "ppu-fetch-trace")]
3425 if let Some(trace) = self.fetch_trace.as_mut() {
3426 trace.push(crate::fetch_trace::FetchRecord {
3427 frame: u32::try_from(self.frame).unwrap_or(u32::MAX),
3428 scanline: self.scanline,
3429 dot: self.dot,
3430 addr: a,
3431 });
3432 }
3433 let val = if a < 0x2000 {
3434 bus.ppu_read(a)
3435 } else {
3436 // Mirror $3000-$3EFF to $2000-$2EFF, unless the cartridge has RAM
3437 // there (v2.9.6: GTROM, UNROM 512 four-screen). Asked only for a
3438 // `$3xxx` address, which rendering never produces.
3439 let nt_addr = if a >= 0x3000 && !bus.nametable_unfolded() {
3440 a - 0x1000
3441 } else {
3442 a
3443 };
3444 if let Some(v) = bus.peek_nametable(nt_addr) {
3445 v
3446 } else {
3447 let off = bus.nametable_address(nt_addr) as usize;
3448 self.ciram[off & 0x07FF]
3449 }
3450 };
3451 // The VRAM data bus latches every read (the rendering fetches drive it);
3452 // a `$2007` read during rendering returns this, not a read at `v`.
3453 {
3454 self.render_data_bus = val;
3455 }
3456 // v2.0.3 (ADR 0030) — TriCNES `FetchPPU`: after the read the multiplexed
3457 // bus's low 8 bits hold the DATA (AD7-0). Crucially, `octal_latch` is NOT
3458 // refreshed here — it retains the ADDRESS low it was loaded with at ALE, so
3459 // a following `$2007`-read ALE-overlap freezes it on this stale data byte
3460 // (the "ALE + Read" corruption; the latch is managed by `ale_splice` /
3461 // `drive_bus`).
3462 val
3463 }
3464
3465 /// W2 ($2007 Stress) — per-dot sprite-tile fetch read cadence (dots
3466 /// 257-320), feeding `render_data_bus` for the deferred `$2007` `PPUDATA`
3467 /// buffer reload. Per `AccuracyCoin` `$2007 Stress` (and `TriCNES`'s per-dot
3468 /// PPU), each 8-dot sprite slot does TWO nametable reads in a row (not
3469 /// NT+AT) then the sprite PT-lo / PT-hi. Both garbage NT reads use the
3470 /// (horizontally-reset) `v` address — the sprite-fetch interval does no
3471 /// coarse-X increment, so it is constant across all 8 slots. Reads land
3472 /// on the slot-local odd dots (1,3,5,7). The PT bytes come from the raw
3473 /// stash captured by `fetch_sprite_tile` (no fresh CHR read, so no new
3474 /// A12/mapper events); the NT reads go through `read_vram`, which latches
3475 /// `render_data_bus` itself.
3476 fn tick_sprite_fetch_read<B: PpuBus>(&mut self, bus: &mut B) {
3477 let local = (self.dot - 257) % 8;
3478 let slot = ((self.dot - 257) / 8) as usize;
3479 match local {
3480 1 | 3 => {
3481 // Slot 0's FIRST garbage NT read straddles the dot-257
3482 // copy-hori boundary: its ALE (dot 257) latched the OLD `v`
3483 // address (`ppudata_spr0_nt_addr`). Every later garbage NT
3484 // read uses the (horizontally reset) live `v`.
3485 let nt = if local == 1 && slot == 0 {
3486 self.ppudata_spr0_nt_addr
3487 } else {
3488 0x2000 | (self.v & 0x0FFF)
3489 };
3490 // `read_vram` latches `render_data_bus` with the value read.
3491 //
3492 // A12 (v2.9.7): on hardware the slot's two garbage reads hold
3493 // A12 low for dots 257-260 of the slot, and the pattern fetch
3494 // raises it at 261. Here the pattern fetch is collapsed into
3495 // `fetch_sprite_tile` at the slot's dot 260, the same dot as the
3496 // second garbage read (local 3), and runs BEFORE it. Reporting
3497 // that read's A12 would lower A12 again after the rise and leave
3498 // it low for eight dots instead of high, which lets an MMC3
3499 // filter count a second rise per slot (mmc3_test 2 "clocked 241
3500 // times" fails). So the first read (local 1) reports the fall,
3501 // the pattern fetch the rise, and the second read stays silent:
3502 // the same edges, one per slot, as the hardware.
3503 let _ = if local == 1 {
3504 self.read_vram(bus, nt)
3505 } else {
3506 self.read_vram_unobserved(bus, nt)
3507 };
3508 }
3509 5 if slot < 8 => self.render_data_bus = self.spr_fetch_lo_raw[slot],
3510 7 if slot < 8 => self.render_data_bus = self.spr_fetch_hi_raw[slot],
3511 _ => {}
3512 }
3513 }
3514
3515 /// Write to PPU memory `$0000-$3EFF`. Mirrors [`Self::read_vram`].
3516 fn write_vram<B: PpuBus>(&mut self, bus: &mut B, addr: u16, value: u8) {
3517 let a = addr & 0x3FFF;
3518 if a < 0x2000 {
3519 bus.ppu_write(a, value);
3520 } else {
3521 let nt_addr = if a >= 0x3000 && !bus.nametable_unfolded() {
3522 a - 0x1000
3523 } else {
3524 a
3525 };
3526 // Give the mapper a chance to absorb the write (ExRAM
3527 // nametables, fill-mode drops, etc.). If declined, write CIRAM.
3528 if !bus.write_nametable(nt_addr, value) {
3529 let off = bus.nametable_address(nt_addr) as usize;
3530 self.ciram[off & 0x07FF] = value;
3531 // v2.3.2 "Lucid" — attribute the byte to the instruction that
3532 // stored it. Recorded here rather than at the CPU-write boundary
3533 // because only this site knows the resolved physical offset: the
3534 // caller wrote `$2007`, and `v` plus the mapper's mirroring is
3535 // what turned that into `off`.
3536 #[cfg(feature = "debug-hooks")]
3537 if let Some(attrib) = self.write_attrib.as_mut() {
3538 attrib.record_ciram(off, self.attrib_pc, self.attrib_cycle, value);
3539 }
3540 }
3541 }
3542 }
3543
3544 /// Read palette RAM. Mirrors:
3545 /// $3F10/$14/$18/$1C → $3F00/$04/$08/$0C
3546 /// anything past $3F1F mirrors back into the 32-byte window.
3547 const fn read_palette(&self, addr: u16) -> u8 {
3548 let idx = palette_index(addr);
3549 // Apply the greyscale mask if PPUMASK bit 0 is set.
3550 let raw = self.palette_ram[idx];
3551 if self.mask.contains(PpuMask::GREYSCALE) {
3552 raw & 0x30
3553 } else {
3554 raw
3555 }
3556 }
3557
3558 // Const-promotable only when `debug-hooks` is off (the attribution branch
3559 // below dereferences a `Box`, which is not const). Allowing the lint keeps
3560 // ONE definition instead of two cfg'd copies of the same three lines.
3561 #[allow(clippy::missing_const_for_fn)]
3562 fn write_palette(&mut self, addr: u16, value: u8) {
3563 let idx = palette_index(addr);
3564 // Palette is 6-bit storage.
3565 self.palette_ram[idx] = value & 0x3F;
3566 // v2.3.2 "Lucid" — record the MASKED value, so the attribution matches
3567 // what a later read returns rather than what the CPU put on the bus.
3568 // `idx` is post-mirroring, so an attribution looked up through `$3F10`
3569 // and through `$3F00` resolves to the same record — correct, since they
3570 // are the same byte.
3571 #[cfg(feature = "debug-hooks")]
3572 if let Some(attrib) = self.write_attrib.as_mut() {
3573 attrib.record_palette(idx, self.attrib_pc, self.attrib_cycle, value & 0x3F);
3574 }
3575 }
3576
3577 /// Re-point the rendering gate to the swept depth, and hand back the
3578 /// 1-dot-delayed value as it stood when the dot began.
3579 ///
3580 /// Called at the TOP of [`Self::tick`], before `rendering_gate` and every
3581 /// other consumer reads `rendering_enabled_delayed` this dot. Depth 1 is
3582 /// the shipped behaviour and leaves the field exactly as the tick-end
3583 /// assignment left it, so the default path is untouched by construction
3584 /// rather than by claim.
3585 ///
3586 /// The return value is the pipeline's stage N-1 and MUST be the value
3587 /// handed to [`Self::render_gate_end_dot`]: at depth >= 2 this function
3588 /// overwrites the field with stage N-2, so reading the field back at
3589 /// tick-end assigns `render_gate_prev2` to itself and the pipeline freezes
3590 /// at its power-on value instead of shifting. That is the defect found in
3591 /// review on #506 and the reason the shift is a named pair rather than two
3592 /// bare assignments a hundred lines apart.
3593 #[cfg(feature = "phi2-write-sweep")]
3594 const fn render_gate_begin_dot(&mut self, depth: u8) -> bool {
3595 let prev1 = self.rendering_enabled_delayed;
3596 match depth {
3597 0 => self.rendering_enabled_delayed = self.mask.rendering_enabled(),
3598 1 => {}
3599 _ => self.rendering_enabled_delayed = self.render_gate_prev2,
3600 }
3601 prev1
3602 }
3603
3604 /// Shift stage N-1 into stage N-2. Called at the END of [`Self::tick`],
3605 /// beside the `rendering_enabled_delayed = rendering` that fills stage N-1,
3606 /// with the value [`Self::render_gate_begin_dot`] returned for this dot.
3607 #[cfg(feature = "phi2-write-sweep")]
3608 const fn render_gate_end_dot(&mut self, prev1: bool) {
3609 self.render_gate_prev2 = prev1;
3610 }
3611
3612 /// Tick exactly one dot.
3613 #[allow(clippy::too_many_lines)] // the per-dot FSM + the ppu-oam-data-bus tick hook
3614 #[allow(clippy::cognitive_complexity)] // + the ppu-sprite-shifter-counter render-toggle branches
3615 pub fn tick<B: PpuBus>(&mut self, bus: &mut B) {
3616 // Advance the dot/scanline FSM first, then handle per-dot events at
3617 // the post-advance position.
3618 self.advance_dot();
3619
3620 // === v2.1.8 A1 — specialized visible-scanline fast dot path ===
3621 //
3622 // The per-dot `tick` FSM below is the emulator's single hottest
3623 // function (`Ppu::tick` ~46% of a representative frame's self-time,
3624 // `docs/performance.md`). The overwhelming majority of its 89,342
3625 // per-frame invocations are visible-scanline BG-render dots whose
3626 // surrounding event/bookkeeping branches are all statically dead —
3627 // no scanline-241 VBL set, no pre-render clear, no OAM-corruption
3628 // edge, no PPUDATA state machine in flight, no `$2006` copy-V or
3629 // PPUMASK write delay pending, rendering stably enabled. This gate
3630 // detects that regime cheaply and, when the (default-OFF) runtime
3631 // knob is on, dispatches to [`Self::tick_visible_render_fast`], which
3632 // runs the *identical* helper sequence with the dead branches pruned.
3633 //
3634 // BYTE-IDENTITY: the fast handler is byte-identical BY CONSTRUCTION —
3635 // it calls the same helpers (`tick_oam_corruption`,
3636 // `tick_sprite_eval_per_dot`, `tick_oam_bus`, `reload_bg_shift_regs`,
3637 // the `ale_drive_*` / `fetch_*` pair, `inc_hori_v`, `inc_vert_v`,
3638 // `emit_pixel`, `shift_bg`) in the same order the general path would
3639 // for a dot satisfying the guard, and executes NONE of the branches
3640 // the guard proves un-taken. The guard is conservative: any doubt
3641 // (delay counters non-zero, corruption armed, cache cold, rendering
3642 // toggling) falls through to the exact path below. Empirically pinned
3643 // bit-for-bit by the differential test
3644 // (`crates/rustynes-test-harness/tests/fast_dotloop_diff.rs`) and the
3645 // full AccuracyCoin / visual-regression / nestest oracle.
3646 //
3647 // Compiled out under `ppu-state-trace` (whose end-of-tick hook must
3648 // observe every dot); under that feature the knob is inert.
3649 // Guard-condition ORDER is tuned to fail fast (short-circuit) for the
3650 // common non-covered dots so the flag costs little when it does not
3651 // apply: the dot-range and rendering-enabled tests eliminate the
3652 // out-of-window and rendering-disabled dots first (a rendering-disabled
3653 // frame — e.g. the `flowing_palette` all-64-colour backdrop-override
3654 // demo — bails at `rendering_enabled()` before the more numerous
3655 // sub-dot-disturbance checks). AND is commutative, so the reorder is
3656 // byte-identity-neutral. NOTE: `cached_visible` is only meaningful once
3657 // the classification cache is warm, so `scanline == flags_cached_scanline`
3658 // MUST be tested before it.
3659 #[cfg(not(feature = "ppu-state-trace"))]
3660 if self.fast_dotloop
3661 && self.dot <= 256
3662 && self.dot >= 1
3663 // Rendering stably enabled: immediate == 1-dot-delayed == previous
3664 // dot's value, so `rendering`, `rendering_gate`, `bg_reload_render`
3665 // and the shift gate all collapse to `true` with no edge to model.
3666 && self.mask.rendering_enabled()
3667 && self.rendering_enabled_delayed
3668 // v2.6.18: the fast body performs its own dot-256 `inc_vert_v`,
3669 // which reads a TWO-dot rendering history, while every other
3670 // rendering term here proves only ONE -- `rendering_enabled_delayed`
3671 // and `prev_rendering_enabled` are both "rendering as of the
3672 // previous dot", assigned together. A mask that turned on at dot
3673 // 255 satisfies all of them at dot 256 with the two-dot view still
3674 // `false`, and the fast body would then increment where the general
3675 // path does not.
3676 //
3677 // HONESTY NOTE, so the next mutation pass does not re-investigate:
3678 // deleting this term is **NOT CAUGHT** by anything in this tree --
3679 // not `fast_dotloop_is_byte_identical_across_corpus`, not the
3680 // AccuracyCoin battery, and not
3681 // `fast_dotloop_is_byte_identical_when_an_enable_lands_beside_dot_256`,
3682 // which was written for exactly this and sweeps twelve power-on
3683 // alignments of the one ROM that writes `$2001` next to dot 256.
3684 // Writes land on CPU-cycle boundaries, so the reachable effect dots
3685 // are three apart, and no stimulus here lands one at 254->255.
3686 //
3687 // It is kept anyway, and that is not stubbornness: the term can only
3688 // make the fast path be taken LESS often, and the general path is
3689 // the reference, so its presence cannot cause a divergence while its
3690 // absence is a latent one waiting for a ROM that writes there.
3691 && self.rendering_enabled_delayed2
3692 && self.prev_rendering_enabled
3693 // Scanline classification cache warm (dot 0 of the line, taken on
3694 // the general path, warms it) AND this is a visible scanline.
3695 && self.scanline == self.flags_cached_scanline
3696 && self.cached_visible
3697 // No sub-dot disturbance in flight.
3698 && self.copy_v_delay == 0
3699 && self.mask_write_delay == 0
3700 && self.ppudata_sm_countdown == 0
3701 && !self.oam_corruption_pending
3702 && !self.oam_corruption_disabled
3703 && !self.oam_corruption_disabled_instant
3704 // Depth-2 sweep only; a no-op constant in the shipped build.
3705 && fast_dot_paths_valid()
3706 {
3707 #[cfg(feature = "ppu-fetch-trace")]
3708 {
3709 self.fast_path_hits = self.fast_path_hits.saturating_add(1);
3710 }
3711 self.tick_visible_render_fast(bus);
3712 return;
3713 }
3714
3715 // === v2.2.3 P2 — specialized IDLE-LINE fast dot path ===
3716 //
3717 // A1 (above) covers visible dots 1..=256 — 61,440 of the 89,342 NTSC
3718 // dots, 68.8%. The remaining 31.2% still walk the whole general body.
3719 // The cheapest slice of that remainder to prove is the **idle line**:
3720 // the post-render line (240) plus every vblank line except the VBL-set
3721 // line 241, i.e. 20 of 262 lines / 6,820 dots per frame.
3722 //
3723 // On such a dot the general path below reduces, provably, to exactly
3724 // three assignments — every other branch is gated on `render_line`,
3725 // `visible`, `pre_render`, `scanline == vblank_start_line()`, or a
3726 // disturbance counter this guard requires to be zero:
3727 //
3728 // * `bg_reload_render = mask.rendering_enabled()` (the
3729 // `mask_write_delay == 0` arm),
3730 // * `prev_rendering_enabled = rendering`,
3731 // * `rendering_enabled_delayed = rendering`.
3732 //
3733 // `tick_idle_line_fast` performs precisely those, in that order, from
3734 // the same single `mask.rendering_enabled()` read — so it is
3735 // byte-identical BY CONSTRUCTION, on the same terms as A1.
3736 //
3737 // The guard requires the classification cache to be WARM for this
3738 // scanline, which is what makes `cached_idle_line` trustworthy: dot 0
3739 // of every line misses the cache and takes the general path (warming
3740 // it), so the fast path serves dots 1..=340 — 340 of each idle line's
3741 // 341 dots.
3742 //
3743 // It also requires the three sub-dot disturbance countdowns to be
3744 // idle. `$2006` (`copy_v_delay`) and `$2001` (`mask_write_delay`) are
3745 // load-bearing: both are perfectly legal during vblank — that is when
3746 // most games issue them — and each has real work to do on landing.
3747 // `ppudata_sm_countdown` is BELT-AND-BRACES: it is armed only under
3748 // `mask.rendering_enabled() && is_render_scanline()` (see the `$2007`
3749 // read handler), so it cannot currently be live on an idle line at all.
3750 // It is tested anyway so the guard's correctness is self-evident from
3751 // the guard itself, rather than resting on an invariant enforced three
3752 // hundred lines away that a future change could quietly break. One
3753 // comparison is a fair price for that.
3754 //
3755 // NOTE for anyone extending this: the three assignments in
3756 // `tick_idle_line_fast` are, given this guard, provably redundant —
3757 // the mask cannot change without a `$2001` write, which arms
3758 // `mask_write_delay` and routes the affected dots through the general
3759 // path, so the values are already correct on every dot the fast path
3760 // serves. Verified empirically: deleting any one of them leaves the
3761 // whole differential suite green. They are kept regardless, because
3762 // "runs the same assignments in the same order" is a claim that can be
3763 // checked by reading twenty lines, whereas "these stores are dead" is a
3764 // reachability argument that must be re-derived every time the guard
3765 // moves. Three stores per idle dot is not worth trading that away.
3766 //
3767 // Compiled out under `ppu-state-trace`, whose end-of-tick hook must
3768 // observe every dot (same treatment as A1).
3769 #[cfg(all(feature = "ppu-idle-line-fast", not(feature = "ppu-state-trace")))]
3770 if self.fast_dotloop
3771 && self.scanline == self.flags_cached_scanline
3772 && self.cached_idle_line
3773 && self.copy_v_delay == 0
3774 && self.mask_write_delay == 0
3775 && self.ppudata_sm_countdown == 0
3776 // Depth-2 sweep only; a no-op constant in the shipped build.
3777 && fast_dot_paths_valid()
3778 {
3779 self.tick_idle_line_fast();
3780 return;
3781 }
3782
3783 // v2.0.3 (ADR 0030, Option 1) — the delayed-`CopyV` landing (`TriCNES`
3784 // `Emulator.cs:1684-1704`). Ticked at the TOP of the dot, BEFORE the fetch
3785 // dispatch, so the `address_bus = v` splice is in place for THIS dot's
3786 // nametable read (the corrupted "Hybrid Addresses" fetch). The phase-0 NT
3787 // ALE of the corrupt group ran on the PREVIOUS dot and already loaded
3788 // `octal_latch` with the one-tile-ahead NT-low, so the armed splice at the
3789 // read below yields `(v & 0x3F00) | octal_latch = $2F19` naturally.
3790 if self.copy_v_delay > 0 {
3791 self.copy_v_delay -= 1;
3792 if self.copy_v_delay == 0 {
3793 self.v = self.t;
3794 self.address_bus = self.v;
3795 // Preserve the `$2006`-write A12 edge (MMC3 timing). It is delayed
3796 // by the countdown vs the flag-off immediate copy, but `$2006`
3797 // writes during active render are rare and A12 during render is
3798 // dominated by the dot-260 sprite fetch, so this stays inside the
3799 // fetch-address-derived-timing budget (verified by the battery).
3800 self.observe_a12(bus);
3801 }
3802 }
3803
3804 // v2.0 Phase 6 (mc-ppu-subpos): track the BG-reload gate. It follows the
3805 // live `self.mask` rendering bit EXCEPT during the analog `$2001`
3806 // write-delay window, where it stays frozen at the prior value (TriCNES
3807 // `PPU_Update2001Delay` -> `PPU_Mask_Show*_Delayed`). Re-syncing to the
3808 // live mask when settled keeps it consistent under a direct mask set
3809 // (unit tests / save-state restore). Done at the top of the dot, before
3810 // the reload runs; the live `self.mask` is unaffected.
3811 if self.mask_write_delay > 0 {
3812 self.mask_write_delay -= 1;
3813 } else {
3814 self.bg_reload_render = self.mask.rendering_enabled();
3815 }
3816
3817 // v1.4.0 Workstream F (F1): the scanline-classification flags are pure
3818 // functions of `self.scanline` + `self.region`, so recompute them only
3819 // when the scanline changes; every other dot reads the cached copies.
3820 // Byte-identical (same values), self-healing on reset / restore.
3821 if self.scanline != self.flags_cached_scanline {
3822 self.cached_visible =
3823 self.scanline >= 0 && self.scanline <= self.region.last_visible_line();
3824 self.cached_pre_render = self.scanline == self.region.prerender_line();
3825 self.cached_render_line = self.cached_visible || self.cached_pre_render;
3826 // v2.2.3 P2 — an "idle" line: neither visible nor pre-render, and not
3827 // the VBL-set line. That is the post-render line (240) plus every
3828 // vblank line except 241 — 20 of the 262 NTSC lines. On such a line
3829 // no dot fetches, renders, evaluates sprites, or raises an event, so
3830 // the whole per-dot body collapses to the rendering-flag bookkeeping
3831 // (see `tick_idle_line_fast`). Cached here with the other
3832 // classification flags because it is the same pure function of
3833 // `scanline` + `region` and shares their `flags_cached_scanline` key.
3834 #[cfg(feature = "ppu-idle-line-fast")]
3835 {
3836 self.cached_idle_line =
3837 !self.cached_render_line && self.scanline != self.region.vblank_start_line();
3838 }
3839 self.flags_cached_scanline = self.scanline;
3840 }
3841 let visible = self.cached_visible;
3842 let pre_render = self.cached_pre_render;
3843 let render_line = self.cached_render_line;
3844 let rendering = self.mask.rendering_enabled();
3845 // v2.6.18 derivation: re-point the delayed gate to the swept depth
3846 // BEFORE `rendering_gate` and every other consumer reads it this dot.
3847 // Depth 1 leaves it exactly as the tick-end assignment left it, so the
3848 // default path is untouched BY CONSTRUCTION rather than by claim.
3849 // The 1-dot-delayed value AS THIS DOT STARTED, captured before the
3850 // re-point below overwrites the field. The tick-end shift needs stage
3851 // N-1 to feed stage N-2, and reading the field back there after a
3852 // depth-2 re-point assigns `render_gate_prev2` to ITSELF — freezing the
3853 // pipeline at its power-on `false` for the whole run instead of
3854 // shifting. Found in review on #506, after the depth sweep had already
3855 // been run: every `lag >= 2` cell measured a permanently-disabled gate
3856 // rather than a two-dot delay. `render_gate_lag_shifts_a_two_dot_pipeline`
3857 // fails without this line.
3858 #[cfg(feature = "phi2-write-sweep")]
3859 let render_gate_prev1 =
3860 self.render_gate_begin_dot(RENDER_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed));
3861 // v2.0 (ae30785): the fetch/shift/sprite-eval pipeline gates on the
3862 // 1-PPU-dot-delayed rendering value under `ppu-sprite-shifter-counter`
3863 // (a mid-scanline `$2001` toggle takes effect one dot later — Stale
3864 // BG/Sprite). Default build = the immediate value (byte-identical).
3865 let rendering_gate = self.rendering_enabled_delayed;
3866 // Stage 2, captured HERE for the same reason stage 1 is: the pipeline
3867 // update below runs BEFORE the dot-event section, so a consumer that
3868 // reads the field after it sees this dot's value rather than the
3869 // delayed one. Reading `self.rendering_enabled_delayed2` at the dot-256
3870 // site gave rendering ONE dot ago and the increment kept firing.
3871 let rendering_gate2 = self.rendering_enabled_delayed2;
3872
3873 // OAM corruption (TriCNES eval-pointer model). The disable edge
3874 // itself is armed by the `$2001` write (see the PPUMASK handler);
3875 // the index is captured against the live secondary-OAM eval
3876 // pointer during the dots 1-64 window (`capture_oam_corruption`,
3877 // called from the sprite-eval FSM); and the actual corruption is
3878 // committed when rendering RE-ENABLES on a render line, or at the
3879 // pre-render line. Here we only retain the unrelated BG-shifter
3880 // fix-up that the prior model happened to share the 1->0 edge with.
3881 if render_line && rendering != self.prev_rendering_enabled && !rendering {
3882 // v2.0 (ppu-sprite-shifter-counter): if rendering is disabled
3883 // mid-pre-fetch (dots 329-336, the SECOND fetch group after the
3884 // dot-329 reload), the in-progress group's pending `<<= 8` would be
3885 // skipped by the now-gated pipeline, freezing the just-reloaded tile
3886 // in the BG shifter's bits 0-7. Complete it ONCE here (one-time on
3887 // the 1->0 edge) so the tile lands in bits 8-15 and surfaces at the
3888 // correct pixel on re-enable (Stale Sprite Shift Regs t5/6).
3889 if (329..=336).contains(&self.dot) {
3890 self.prefetch_shift_bg_regs();
3891 }
3892 }
3893 // Commit pending OAM corruption at the START of the pre-render line
3894 // (TriCNES handles this via the dots 1-64 eval path on the
3895 // pre-render line; the dot-0 hook covers the case where rendering
3896 // was re-enabled during VBlank and stays on into pre-render).
3897 // Per TriCNES `CorruptOAM`, the corruption applies on the first
3898 // rendered dot once rendering is (re-)enabled.
3899 if self.scanline == self.region.prerender_line()
3900 && self.dot == 0
3901 && rendering
3902 && self.oam_corruption_pending
3903 {
3904 self.process_oam_corruption();
3905 }
3906 // OAM corruption (TriCNES eval-pointer model): maintain the
3907 // `OAM2Address` analogue across the dots 1-64 clear window, capture
3908 // the corruption index at the disable edge, and commit on re-enable.
3909 // Driven every render-line dot independent of the rendering gate so
3910 // the disable edge is observed even though the sprite-eval FSM below
3911 // stops once rendering is gated off.
3912 if render_line {
3913 self.tick_oam_corruption(rendering);
3914 }
3915 self.prev_rendering_enabled = rendering;
3916 // v2.0 (ae30785): update the 1-dot-delayed copy AFTER this dot's gate
3917 // read above, so the next dot sees the delayed value.
3918 {
3919 #[cfg(feature = "phi2-write-sweep")]
3920 {
3921 self.render_gate_end_dot(render_gate_prev1);
3922 // Newest first: stage 0 is "one dot ago" on the next dot.
3923 self.sweep_mask_history.rotate_right(1);
3924 self.sweep_mask_history[0] = self.mask;
3925 }
3926 // Stage 2 takes stage 1 as it stood when this dot BEGAN, which is
3927 // exactly the `rendering_gate` local. Ordering is load-bearing in
3928 // the same way `render_gate_end_dot` is: reading the field back
3929 // here instead would assign stage 1 to stage 2 after stage 1 had
3930 // already been overwritten, freezing the pipeline -- the #506
3931 // defect in a second place.
3932 self.rendering_enabled_delayed2 = rendering_gate;
3933 self.rendering_enabled_delayed = rendering;
3934 }
3935
3936 // === Dot-1 / dot-0 events ===
3937 // VBL flag is set at scanline 241 dot 1 per nesdev wiki. The
3938 // PPU's /NMI line is pulled low one PPU clock later (dot 2),
3939 // matching the behavior blargg's `ppu_vbl_nmi/05-nmi_timing` and
3940 // `08-nmi_off_timing` were calibrated to: a /NMI assertion-edge
3941 // sample ~6-7 PPU clocks after VBL set, given how our bus
3942 // interleaves CPU bus accesses before the 3-PPU-tick `on_cpu_cycle`
3943 // hook (vs. real hardware's mid-cycle phi1 access).
3944 #[cfg(not(feature = "phi2-write-sweep"))]
3945 let vbl_set_dot = 1u16;
3946 // v2.6.18 condition-2 diagnostic. NOT a proposed fix: nesdev states the
3947 // flag sets at scanline 241 dot 1, and this knob exists only to ask
3948 // whether the six NMI entries' failure at a later ACCESS dot is a pure
3949 // one-dot RELATIVE shift between the `$2002` read and VBL-set, or
3950 // something else. If shifting VBL by the same dot restores all six, the
3951 // disagreement is about CPU/PPU alignment rather than about the read.
3952 #[cfg(feature = "phi2-write-sweep")]
3953 let vbl_set_dot = u16::from(VBL_SET_DOT.load(core::sync::atomic::Ordering::Relaxed));
3954 if self.scanline == self.region.vblank_start_line()
3955 && self.dot == vbl_set_dot
3956 && !self.suppress_vbl_this_frame
3957 {
3958 self.status.insert(PpuStatus::VBLANK);
3959 // Inform the mapper that we have entered VBL — MMC5 uses this
3960 // to clear its in-frame flag.
3961 bus.notify_vblank();
3962 // R2 (mc-r1-substrate): /NMI is asserted on the SAME dot as
3963 // VBL-set when NMI_ENABLE is set (Mesen2 `NesPpu.cpp:1339-1343`).
3964 // On R1's on-time access the CPU read at dot 1 lands AFTER
3965 // VBL+NMI set, so the v1.x `dot==3` lag-comp band-aid (below) is
3966 // obsolete and disabled.
3967 if self.ctrl.contains(PpuCtrl::NMI_ENABLE) {
3968 self.nmi_line = true;
3969 }
3970 }
3971 // v1.x default path: /NMI raised at dot 3 to compensate for the
3972 // lockstep PPU running ~8 mc late. Disabled under the on-time R1
3973 // substrate (R2 raises /NMI at dot 1, above).
3974 if pre_render && self.dot == 1 {
3975 self.status.remove(
3976 PpuStatus::VBLANK
3977 .union(PpuStatus::SPRITE_ZERO_HIT)
3978 .union(PpuStatus::SPRITE_OVERFLOW),
3979 );
3980 self.nmi_line = false;
3981 self.suppress_vbl_this_frame = false;
3982 }
3983
3984 // Notify the mapper that a rendered scanline has started. We fire
3985 // this on dot 0 of every visible line and the pre-render line,
3986 // before any pattern/attribute fetches happen. MMC5 uses this to
3987 // tick its scanline IRQ counter (which conceptually fires at PPU
3988 // cycle ~4 of each rendered line — close enough for v0). Other
3989 // mappers default to no-op.
3990 if render_line && self.dot == 0 {
3991 bus.notify_scanline_start();
3992 // T-MMC3-BG-A12, rule 2: a visible line's dot 0 "appears to be
3993 // the same CHR address that is later used to fetch the low
3994 // background tile byte starting at dot 5" (`NESdev` PPU
3995 // rendering, "Cycle 0"), so with the background at `$1000` A12 is
3996 // already high here. Not on the pre-render line, which the page
3997 // does not extend the statement to and where giving it the rule
3998 // failed blargg `4-scanline_timing` at sub-test 8; and not on the
3999 // dot the odd-frame skip replaced (`dot0_replaced`), which is the
4000 // dummy nametable fetch's tick. The rule is what closes sub-test 12
4001 // (removing it returns both 4-scanline ROMs there); the second
4002 // exception is documented behaviour that no ROM in the corpus
4003 // reaches, pinned by the unit test
4004 // `scanline_0_dot_0_drives_bg_chr_only_when_the_skip_did_not_replace_it`.
4005 if self.mask.rendering_enabled()
4006 && !self.dot0_replaced
4007 && self.scanline != self.region.prerender_line()
4008 {
4009 let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4010 self.observe_a12_addr(bus, bg_table);
4011 }
4012 self.dot0_replaced = false;
4013 }
4014
4015 // === Background rendering pipeline (visible + pre-render lines) ===
4016 // v2.6.18 condition-4: when `SCROLL_GATE_LAG` is armed, the dot-256
4017 // vertical increment is evaluated HERE on its own `$2001` view.
4018 //
4019 // This is the consumer `Frozen OAM2 Increment` actually turns on. The
4020 // ROM enables rendering ON dot 256 and states outright that "the PPU's
4021 // vertical scroll is NOT incremented"; our enable lands a dot early, so
4022 // `inc_vert_v` fires and `v` ends at fine-Y 3 where the test needs 2 --
4023 // no opaque background pixel under the sprite, no sprite-zero hit, and
4024 // the entry fails for a reason that has nothing to do with OAM2.
4025 // `RENDER_GATE_LAG = 2` fixes it and breaks `Stale Sprite Shift Regs`,
4026 // because that depth moves every other consumer too.
4027 // v2.6.18: the dot-256 vertical increment reads rendering as of TWO
4028 // dots ago, not the shared one-dot gate -- see
4029 // `rendering_enabled_delayed2`. Hoisted out of the `render_line &&
4030 // rendering_gate` block so the shared gate cannot decide the question
4031 // before the deeper one is consulted; conjoining the two would put the
4032 // DISABLE edge back on the shallower gate, which is the edge
4033 // `Frozen OAM2 Increment` test 4 measures.
4034 if render_line
4035 && self.dot == 256
4036 && Self::scroll_gate_follows_render_gate()
4037 && rendering_gate2
4038 {
4039 self.inc_vert_v();
4040 }
4041 #[cfg(feature = "phi2-write-sweep")]
4042 if render_line && self.dot == 256 && !Self::scroll_gate_follows_render_gate() {
4043 let depth = SCROLL_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) as usize;
4044 let m = if depth == 0 {
4045 self.mask
4046 } else {
4047 self.sweep_mask_history[(depth - 1).min(self.sweep_mask_history.len() - 1)]
4048 };
4049 if m.rendering_enabled() {
4050 self.inc_vert_v();
4051 }
4052 }
4053 // v2.6.18: when `SPRITE_REARM_LAG` is armed the dot-339 re-arm is
4054 // evaluated HERE, outside the shared `rendering_gate` block, so it can
4055 // see a rendering value the shared gate does not. Hoisted rather than
4056 // gated in place because the enclosing block would otherwise decide the
4057 // question before the knob is consulted.
4058 #[cfg(feature = "phi2-write-sweep")]
4059 if render_line && self.dot == 339 && !Self::sprite_rearm_follows_render_gate() {
4060 let depth = SPRITE_REARM_LAG.load(core::sync::atomic::Ordering::Relaxed) as usize;
4061 let m = if depth == 0 {
4062 self.mask
4063 } else {
4064 self.sweep_mask_history[(depth - 1).min(self.sweep_mask_history.len() - 1)]
4065 };
4066 if m.rendering_enabled() {
4067 for i in 0..self.spr_count as usize {
4068 self.spr_halted[i] = false;
4069 }
4070 }
4071 }
4072 if render_line && rendering_gate {
4073 // Sprite evaluation: per-PPU-dot FSM matching real-hardware
4074 // behavior (cycles 1-64 secondary-OAM clear, 65-256 alternating
4075 // odd/even read/write with the documented buggy `n+m`
4076 // overflow-detection increment). Visible scanlines evaluate
4077 // for the next visible scanline; the pre-render line evaluates
4078 // for scanline 0. Without the pre-render eval, secondary OAM
4079 // from the last visible scanline would leak into pre-render's
4080 // dummy sprite tile fetches, causing wrong A12 emissions and
4081 // incorrect sprite-zero state.
4082 if visible || pre_render {
4083 self.tick_sprite_eval_per_dot();
4084 }
4085 // v2.0 Tier 1.2: drive the isolated OAM-data-bus model on visible
4086 // scanlines when rendering, so a CPU $2004 read mid-frame observes
4087 // the sprite-eval / load data bus (AccuracyCoin `$2004 Stress`).
4088 // Side-effect-free w.r.t. the rendering FSM above.
4089 // v2.6.18: the OAM2 machinery's effective gate is a CONJUNCTION --
4090 // this test AND the enclosing `render_line && rendering_gate`. That
4091 // matters and is easy to state wrongly: for an AND of two delayed
4092 // views of one signal, the DISABLE edge fires at the shallower depth
4093 // and the RE-ENABLE edge at the deeper one. So today
4094 // `OAM2_GATE_LAG = 0` owns the disable edge and the 1-dot
4095 // `rendering_enabled_delayed` owns the re-enable edge -- the two
4096 // knobs each own ONE edge, which is why neither alone can place the
4097 // window and why `Frozen OAM2 Increment` moves only when
4098 // `RENDER_GATE_LAG` does.
4099 //
4100 // TriCNES has no such nesting: `PPU_Render_SpriteEvaluation()` is
4101 // called unconditionally and each block tests exactly one mask view
4102 // (`Emulator.cs:2073`, `:2076-2082`).
4103 //
4104 // `OAM2_GATE_LAG` makes this test's depth swept rather than assumed;
4105 // 0 is the shipped live-mask read of THIS term, not of the gate.
4106 // v2.6.18: this used to conjoin the LIVE mask, which put the OAM2
4107 // counter's DISABLE edge a dot ahead of every other consumer --
4108 // `min(r, d)` for a conjunction of two delayed views. The dot-339
4109 // reset then got skipped by a write whose effect lands DURING 339,
4110 // leaving the "OAM2 Overflowed" flag raised and producing the
4111 // sprite-zero hit `Frozen OAM2 Increment` test 4 exists to forbid.
4112 // The counter rides the shared one-dot gate like everything else,
4113 // which the enclosing block has already applied.
4114 if visible && self.oam2_gate_rendering() {
4115 self.tick_oam_bus();
4116 }
4117 // Sprite tile fetch + A12 emission. Real hardware spreads the
4118 // 8 sprite slots' pattern fetches across cycles 257..=320 — for
4119 // each slot, garbage NT bytes at +1/+3, sprite pattern lo at
4120 // +5/+6, sprite pattern hi at +7/+8. We collapse that to per-
4121 // slot emission at dot 260 for slot 0, 268 for slot 1, …, 316
4122 // for slot 7. This is the canonical "MMC3 IRQ at PPU dot 260"
4123 // timing — the first A12 rise to the sprite pattern table
4124 // happens here for standard pattern-table layout (BG=$0000,
4125 // sprites=$1000), per `docs/mappers.md` §MMC3 → IRQ counter
4126 // mechanism.
4127 //
4128 // CRITICAL for MMC3: even unused sprite slots ALWAYS perform
4129 // the dummy sprite-pattern fetch on real hardware (using the
4130 // cleared secondary-OAM tile $FF), so A12 toggles into the
4131 // sprite pattern table once per scanline regardless of how
4132 // many real sprites are visible. This must run on both
4133 // visible scanlines and the pre-render line — pre-render
4134 // sprite fetches are for scanline 0's sprites and contribute
4135 // the 241st A12 rising edge per frame (240 visible + 1
4136 // pre-render) that MMC3's IRQ counter expects.
4137 if (260..=316).contains(&self.dot) {
4138 let phase = self.dot.wrapping_sub(260);
4139 if phase.trailing_zeros() >= 3 {
4140 let slot = (phase >> 3) as usize;
4141 self.fetch_sprite_tile(bus, slot);
4142 }
4143 }
4144
4145 // OAMADDR reset: per nesdev wiki "PPU registers" §OAMADDR,
4146 // "OAMADDR is set to 0 during each of ticks 257-320 (the
4147 // sprite tile loading interval) of the pre-render and visible
4148 // scanlines." This is the hardware behaviour that lets games
4149 // STX $4014 their OAM-staging page after rendering without
4150 // having to remember to STA $2003 #0 first. Required for
4151 // AccuracyCoin TEST_Sprite0Hit_Behavior subtest 1 (which
4152 // relies on the prior test-runner's OAMADDR perturbation
4153 // being washed away by the previous frame's rendering).
4154 if (257..=320).contains(&self.dot) {
4155 self.oam_addr = 0;
4156 }
4157
4158 // v2.0 (ppu-sprite-shifter-counter): at dot 339 of a render line,
4159 // re-arm the LOADED sprite counters (the `spr_count` slots fetched
4160 // for the next scanline) to "counting". This whole block is gated on
4161 // `rendering_gate`, so a render-disable across dot 339 leaves the
4162 // loaded slots halted — a reloaded-but-halted counter draws
4163 // immediately on re-enable (Stale Sprite Shift Regs t5/6). Slots
4164 // beyond `spr_count` retain their halted latch.
4165 if self.dot == 339 && Self::sprite_rearm_follows_render_gate() {
4166 for i in 0..self.spr_count as usize {
4167 self.spr_halted[i] = false;
4168 }
4169 }
4170
4171 // BG fetches happen at dots 1..=256 and 321..=336.
4172 //
4173 // CYCLE-PRECISE BG PIPELINE (Mesen2-faithful, fixes Cascade A
4174 // VerifySpriteZeroHits step-2 off-by-one):
4175 //
4176 // Per nesdev wiki "PPU rendering": "The shifters are reloaded
4177 // during ticks 9, 17, 25, ..., 257." Per Mesen2
4178 // `Core/NES/NesPpu.cpp::LoadTileInfo()` (line 667), the reload
4179 // is `case 1` of `(_cycle & 0x07)` — i.e., phase 0 of each
4180 // 8-cycle group, OR'ing the latched LowByte/HighByte into the
4181 // shifter's low 8 bits. The PRIOR group's 8 shifts (one per
4182 // cycle of dots 1..=256 of a visible scanline) leave bits 0-7
4183 // zeroed, so the OR is effectively an overwrite. The pre-fetch
4184 // groups at dots 321..=336 do NOT shift per-cycle; instead
4185 // Mesen2 substitutes a `<<= 8` at phase 7 (dots 328 and 336)
4186 // to clear bits 0-7 for the next reload.
4187 //
4188 // The matching pixel-emit + shift ordering is: emit_pixel reads
4189 // bit (15 - fine_x) FIRST, then shift_bg runs LAST. This is the
4190 // critical change from the prior (off-by-one) implementation
4191 // that shifted BEFORE emit and reloaded at phase 7 (cycle 8).
4192 // See `docs/audit/cascade-a-investigation-2026-05-19.md` for
4193 // the empirical analysis and the per-cycle trace of
4194 // VerifySpriteZeroHits step 2 demonstrating why this is the
4195 // load-bearing change.
4196 let in_bg_fetch = (1..=256).contains(&self.dot) || (321..=336).contains(&self.dot);
4197 if in_bg_fetch {
4198 let phase = (self.dot.wrapping_sub(1)) & 7;
4199 // Phase 0 (cycles 1, 9, 17, ..., 249, 321, 329): reload the
4200 // shifter's low 8 bits from the latches written by the
4201 // PRIOR fetch group. Implementation note: `reload_bg_shift_regs`
4202 // overwrites bits 0-7 via `(shift & 0xFF00) | latch`; this
4203 // matches Mesen2's `|=` because the 8 shifts since the prior
4204 // reload guarantee bits 0-7 are zero before the OR.
4205 if phase == 0 {
4206 // v2.0 Phase 6 (mc-ppu-subpos): the reload is additionally
4207 // gated on the analog-delayed `$2001` value, so a render
4208 // re-enable lands the reload `MASK_WRITE_DELAY` dots after
4209 // the shifter has already resumed advancing -> one reload is
4210 // SKIPPED and the serial-in '1's survive (BG Serial In).
4211 // When stable, `bg_reload_render` == the live rendering gate,
4212 // so this is byte-identical.
4213 let reload_gate = self.bg_reload_render;
4214 if reload_gate {
4215 self.reload_bg_shift_regs();
4216 }
4217 }
4218
4219 // v2.0.3 (ADR 0030, Option 1) — the ALE (address-latch-enable)
4220 // half of each 2-cycle VRAM access. On the EVEN dot of each pair
4221 // (phases 0/2/4/6, i.e. one dot BEFORE the corresponding read at
4222 // phases 1/3/5/7) the PPU drives the full 14-bit fetch address onto
4223 // `address_bus` and captures its low byte into `octal_latch`; the
4224 // read half (the existing `fetch_*` below) reads via the splice
4225 // `(address_bus & 0x3F00) | octal_latch`. For a coherent fetch the
4226 // splice returns the intended address, so this is behavior-neutral;
4227 // the split exists so a later phase's `$2006`/`$2007` corruption can
4228 // desync the two halves naturally.
4229 match phase {
4230 // Phase 0 (NT ALE): drive the PLAIN nametable address
4231 // (`0x2000 | (v & 0x0FFF)`) and load `octal_latch` with its low
4232 // byte. This is a TRUE two-dot ALE for the common (non-MMC5-
4233 // split) case, so the latch naturally carries the NT-low the
4234 // "Hybrid Addresses" corruption needs. The MMC5 vertical-split
4235 // query stays at the read dot (phase 1, `fetch_nt`) because its
4236 // `split_chr_bank_latch` side effect is mapper-observable; when
4237 // that query turns out to be split-active, `fetch_nt` disarms
4238 // this ALE and reads `split.nt_addr` co-located instead, so
4239 // split rendering is byte-identical (no phase-0 mapper query).
4240 0 => self.ale_drive_nt(),
4241 2 => self.ale_drive_at(),
4242 4 => self.ale_drive_bg_lo(),
4243 6 => self.ale_drive_bg_hi(),
4244 _ => {}
4245 }
4246
4247 // 8-cycle fetch group: dot phase = (dot - 1) & 7
4248 // 1 -> NT byte fetch (cycle 2 of group)
4249 // 3 -> AT byte fetch (cycle 4 of group)
4250 // 5 -> BG-low fetch (cycle 6 of group)
4251 // 7 -> BG-high fetch +
4252 // coarse-X increment (cycle 8 of group)
4253 match phase {
4254 1 => self.fetch_nt(bus),
4255 3 => self.fetch_at(bus),
4256 5 => self.fetch_bg_lo(bus),
4257 7 => self.fetch_bg_hi(bus),
4258 _ => {}
4259 }
4260 self.observe_bg_a12_lead(bus, phase);
4261 if phase == 7 {
4262 self.inc_hori_v();
4263 // Pre-fetch region only (dots 328 and 336): explicit
4264 // `<<= 8` to substitute for the missing per-cycle
4265 // shifts during pre-fetch. Per Mesen2
4266 // `ProcessScanlineImpl()` lines 941-944.
4267 if (321..=336).contains(&self.dot) {
4268 self.prefetch_shift_bg_regs();
4269 }
4270 }
4271 }
4272 // Dot 257: the LAST shift-register reload of the visible region
4273 // consumes the latches from the dots-249..=256 fetch group. Dot
4274 // 257 is outside the dots 1..=256 `in_bg_fetch` range above (it
4275 // belongs to the sprite-tile-fetch window 257..=320), but per
4276 // Mesen2's `_cycle <= 256` LoadTileInfo cycle range, this reload
4277 // actually never fires in Mesen2 either — the dot-256 fetch's
4278 // bg_lo/bg_hi latches are consumed by the dot-321 reload (which
4279 // OR's them in, then the dot-328 `<<= 8` shifts them up to
4280 // bits 8-15). So for our model, the dot-256 fetch's latches
4281 // similarly persist past dot 256 into dot 321's reload.
4282 // (Intentionally no dot-257 reload here.)
4283
4284 // Cycle 256: vertical-V increment -- HOISTED to the top of the
4285 // dot-event section (search `rendering_enabled_delayed2`), because
4286 // it reads a rendering gate one dot deeper than the shared one and
4287 // must therefore not sit inside the shared gate's block.
4288 // Cycle 257: copy hori(t) -> hori(v).
4289 if self.dot == 257 {
4290 // W2 ($2007 Stress): the FIRST garbage NT read of sprite slot
4291 // 0 has its ALE at dot 257, BEFORE hori(v) is reset — so its
4292 // ADDRESS uses the OLD `v` (coarse-x wrapped past the row's
4293 // last column) even though the data lands at dot 258. Latch
4294 // the address here; `tick_sprite_fetch_read` uses it for the
4295 // slot-0 first read (key idx 128 = `02`). The later garbage
4296 // NT reads (ALE dots 259+) all use the reset `v`.
4297 {
4298 self.ppudata_spr0_nt_addr = 0x2000 | (self.v & 0x0FFF);
4299 }
4300 self.copy_hori_t_to_v();
4301 }
4302 // Pre-render cycles 280..=304: copy vert(t) -> vert(v).
4303 if pre_render && (280..=304).contains(&self.dot) {
4304 self.copy_vert_t_to_v();
4305 }
4306 // Sprite tile fetch happens in fetch_sprite_tile (dots 260, 268, ..., 316).
4307 // Cycles 337..=340: 2 garbage NT fetches (no-op except A12).
4308 if (337..=340).contains(&self.dot) && (self.dot & 1) == 1 {
4309 self.fetch_nt(bus);
4310 }
4311
4312 // W2 ($2007 Stress): the per-dot sprite-fetch read cadence (dots
4313 // 257-320) feeds `render_data_bus` (NT,NT,PT-lo,PT-hi per 8-dot
4314 // slot) so a deferred `$2007` buffer reload landing in HBlank
4315 // captures the byte the sprite fetch drove on the VRAM bus. The
4316 // real fetch is the collapsed `fetch_sprite_tile` above; this is
4317 // side-effect-free w.r.t. rendering and A12 (the garbage NT reads
4318 // are CIRAM/nametable reads; the PT bytes come from the stash).
4319 if (257..=320).contains(&self.dot) {
4320 self.tick_sprite_fetch_read(bus);
4321 }
4322 }
4323
4324 // W2 ($2007 Stress): the PPUDATA state machine's read step (PD_RB) +
4325 // TStep. A `$2007` read during rendering armed this countdown; one
4326 // tick per PPU dot (unconditionally, so a mid-flight rendering
4327 // disable cannot wedge it), and at 0:
4328 // 1. `data_buffer` latches the value the FETCH cadence drove on the
4329 // VRAM bus at THIS dot (`render_data_bus`, freshly set by the
4330 // fetch dispatch above). Latched bus value only — never a fresh
4331 // VRAM read (zero new A12/mapper events).
4332 // 2. The deferred v-glitch increment (the TStep) fires AFTER the
4333 // reload, per TriCNES `PPU_DATA_StateMachine_Half` — so every
4334 // fetch in the read-to-reload window used the OLD `v`. The
4335 // rendering-vs-blanking choice uses the state AT the TStep dot
4336 // (mirrors TriCNES's `PPU_2007_BLNK_Latch`).
4337 if self.ppudata_sm_countdown > 0 {
4338 self.ppudata_sm_countdown -= 1;
4339 if self.ppudata_sm_countdown == 0 {
4340 self.data_buffer = self.render_data_bus;
4341 // v2.0.2 (ADR 0030) — "ALE + Read": the $2007 read's PPUDATA
4342 // state machine takes 3 PPU cycles to the background cadence's 2,
4343 // so on the reload dot its ALE overlaps a fetch's read. Both ALE
4344 // and READ are asserted, so the octal latch is frozen on the
4345 // read's DATA byte (`render_data_bus`) instead of the next
4346 // Pattern-Address-Register low byte. The next pattern fetch then
4347 // reads `{PAR high 6}:{stale data low}` (`$0F03` -> `$0FFF`).
4348 if self.mask.rendering_enabled() && self.is_render_scanline() {
4349 // Freeze the latch on the read's DATA byte here. The frozen byte
4350 // is then carried by `drive_bus` (which suppresses the next
4351 // pattern ALE's latch reload) and consumed by the pattern read's
4352 // natural `ale_splice`, so the next pattern fetch reads
4353 // `(PAR high 6):(stale $FF) = $0FFF` with no explicit splice.
4354 self.octal_latch = self.render_data_bus;
4355 self.pattern_latch_stale = true;
4356 octal_trace::push(
4357 octal_trace::K_SMLAND,
4358 self.frame,
4359 self.scanline,
4360 self.dot,
4361 u32::from(self.render_data_bus),
4362 );
4363 }
4364 if self.ppudata_v_inc_pending {
4365 self.ppudata_v_inc_pending = false;
4366 if self.mask.rendering_enabled() && self.is_render_scanline() {
4367 self.inc_hori_v();
4368 self.inc_vert_v();
4369 } else {
4370 let inc = if self.ctrl.contains(PpuCtrl::VRAM_INCREMENT_32) {
4371 32
4372 } else {
4373 1
4374 };
4375 self.v = self.v.wrapping_add(inc) & 0x7FFF;
4376 }
4377 // The v change can move A12 (mirrors the read-time path).
4378 self.observe_a12(bus);
4379 }
4380 }
4381 }
4382
4383 // === Pixel emission (visible scanlines, dots 1..=256) ===
4384 // Per Mesen2 `ProcessScanlineImpl()` (lines 881-884), the
4385 // canonical order is: LoadTileInfo (reload at phase 0, fetches at
4386 // phases 1/3/5/7) THEN DrawPixel THEN ShiftTileRegisters. The
4387 // shift-AFTER-emit ordering is the load-bearing other half of the
4388 // Cascade A BG-pipeline fix: emit reads bit (15 - fine_x) of the
4389 // shifter at its CURRENT state (post-reload, pre-shift), then the
4390 // shift advances the register for the next emit. Combined with
4391 // the phase-0 reload above, this places the newly-fetched tile's
4392 // MSB at shift-register bit 15 (the emit read point) at exactly
4393 // PPU dot 9 of each 8-cycle group = pixel column 8.
4394 if visible && (1..=256).contains(&self.dot) {
4395 self.emit_pixel();
4396 // v2.0 (ae30785): the BG shifter advances on the 1-dot-delayed
4397 // rendering gate (under the feature), so a precisely-timed `$2001`
4398 // toggle that skips the reload while rendering stays enabled
4399 // surfaces the serial-in (BG Serial In / Stale BG Shift).
4400 //
4401 // v2.0 Phase 6 (mc-ppu-subpos): TriCNES drives the SHIFT off the
4402 // IMMEDIATE PPUMASK (`_EmulateHalfPPU` -> `PPU_UpdateBackground
4403 // ShiftRegisters`, gated on `PPU_Mask_Show*` not `*_Delayed`) while
4404 // the fetch+RELOAD ride the 1-dot-DELAYED mask
4405 // (`PPU_Render_ShiftRegistersAndBitPlanes`, gated on `*_Delayed`).
4406 // So on a render re-enable edge there is a 1-dot window where the
4407 // shifter advances (injecting the serial-in '1') but the reload is
4408 // still gated off -> a single reload is SKIPPED while shifting
4409 // continues, surfacing the accumulated serial-in '1's at the output
4410 // (AccuracyCoin "BG Serial In"). When no mid-scanline toggle is in
4411 // flight immediate == delayed, so normal rendering is byte-identical.
4412 let shift_gate = render_line && rendering;
4413 if shift_gate {
4414 self.shift_bg();
4415 }
4416 }
4417
4418 // === Per-PPU-dot state-trace recording (Session-10) ===
4419 //
4420 // Gated on the `ppu-state-trace` cargo feature so the
4421 // default build's hot tick path is byte-identical to
4422 // pre-Session-10. The hook reads `self`'s state AFTER
4423 // all this dot's effects have applied, so the captured
4424 // record reflects "PPU state at the end of dot
4425 // (scanline, dot)". It NEVER writes to PPU state — the
4426 // determinism contract is preserved.
4427 //
4428 // See `docs/adr/0005-ppu-state-trace.md`.
4429 #[cfg(feature = "ppu-state-trace")]
4430 if self.state_trace.is_some() {
4431 let rec = self.build_state_record();
4432 if let Some(t) = self.state_trace.as_mut() {
4433 t.maybe_push(rec);
4434 }
4435 }
4436 }
4437
4438 /// v2.1.8 A1 — the specialized straight-line body for a "clean" visible
4439 /// BG-render dot: a visible scanline, `dot` in `1..=256`, rendering stably
4440 /// enabled, and no sub-dot disturbance in flight. Dispatched from
4441 /// [`Self::tick`] behind the [`Self::fast_dotloop`] guard (default ON
4442 /// since v2.2.3).
4443 ///
4444 /// This executes the *exact same* helper sequence the general per-dot path
4445 /// runs for such a dot — in the same order — with every event and
4446 /// bookkeeping branch the guard proves un-taken (VBL/NMI set/clear, the
4447 /// pre-render vertical reload, sprite-tile fetch dots 260..=316, the
4448 /// OAMADDR-reset window, the dot-257 hori-copy, the PPUDATA state machine,
4449 /// the OAM-corruption commit, the odd-frame skip) elided. It is therefore
4450 /// byte-identical to the general path by construction, and is additionally
4451 /// pinned bit-for-bit by the differential test + the full oracle. See the
4452 /// extensive rationale at the dispatch site in [`Self::tick`].
4453 /// v2.2.3 P2 — the specialized straight-line body for an **idle-line** dot:
4454 /// a post-render or vblank line other than the VBL-set line, with no
4455 /// sub-dot disturbance in flight. Dispatched from [`Self::tick`] behind the
4456 /// [`Self::fast_dotloop`] guard.
4457 ///
4458 /// An idle line issues no VRAM fetch, emits no pixel, runs no sprite
4459 /// evaluation, clocks no shifter, and raises no VBL / NMI / A12 /
4460 /// scanline-start event. Walking the general per-dot body for it therefore
4461 /// evaluates ~30 predicates to perform three assignments. This is those
4462 /// three assignments, in the general path's order, derived from one
4463 /// `mask.rendering_enabled()` read exactly as it does:
4464 ///
4465 /// 1. `bg_reload_render` — the general path's `mask_write_delay` `else`
4466 /// arm. The guard proves the countdown is zero, so that arm is the one
4467 /// taken.
4468 /// 2. `prev_rendering_enabled` — assigned unconditionally there.
4469 /// 3. `rendering_enabled_delayed` — likewise, and deliberately AFTER (2):
4470 /// the general path updates the 1-dot-delayed copy last so the *next*
4471 /// dot observes it, and reordering here would shift a mid-vblank
4472 /// `$2001` toggle by one dot.
4473 ///
4474 /// Byte-identical by construction, and pinned bit-for-bit by
4475 /// `fast_dotloop_diff` (which compares whole frames — every idle dot
4476 /// included — and by `idle_line_fast_path_matches_exact_under_vblank_io`,
4477 /// which drives `$2000`/`$2001`/`$2006`/`$2007` during vblank so the
4478 /// guard's fall-through arms are exercised rather than assumed).
4479 #[cfg(feature = "ppu-idle-line-fast")]
4480 #[inline]
4481 const fn tick_idle_line_fast(&mut self) {
4482 let rendering = self.mask.rendering_enabled();
4483 self.bg_reload_render = rendering;
4484 self.prev_rendering_enabled = rendering;
4485 // Stage 2 before stage 1, for the same ordering reason as the general
4486 // path: stage 1's PREVIOUS value is what stage 2 takes.
4487 self.rendering_enabled_delayed2 = self.rendering_enabled_delayed;
4488 self.rendering_enabled_delayed = rendering;
4489 }
4490
4491 // Dead under `ppu-state-trace`, and legitimately so: the dispatch above is
4492 // `#[cfg(not(feature = "ppu-state-trace"))]`, because the trace hook must
4493 // observe EVERY dot and the fast path exists precisely to skip per-dot work.
4494 // Live by default, dead under one feature -- the one shape that earns an
4495 // `allow` rather than a deletion. Scoped to that feature so the attribute
4496 // cannot silently start suppressing a real finding in the default build.
4497 #[cfg_attr(feature = "ppu-state-trace", allow(dead_code))]
4498 #[inline]
4499 fn tick_visible_render_fast<B: PpuBus>(&mut self, bus: &mut B) {
4500 let dot = self.dot;
4501
4502 // General-path top: with `mask_write_delay == 0` (guard) the BG-reload
4503 // gate follows the stably-enabled rendering bit.
4504 self.bg_reload_render = true;
4505
4506 // OAM-corruption pointer bookkeeping. The guard proved nothing is
4507 // armed/pending/disabled, so this only maintains `oam2_addr` across the
4508 // dots 1..=64 secondary-OAM clear window (and is a two-compare no-op for
4509 // dots 65..=256) — exactly what the general path's
4510 // `if render_line { tick_oam_corruption(rendering) }` does here.
4511 self.tick_oam_corruption(true);
4512
4513 // Rendering-edge bookkeeping the NEXT dot's gate consumes. The general
4514 // path assigns all three every dot (`delayed2 <- delayed <- rendering`,
4515 // `prev <- rendering`); here the guard in `tick` has already required
4516 // `prev_rendering_enabled`, `rendering_enabled_delayed`,
4517 // `rendering_enabled_delayed2` and `mask.rendering_enabled()`, so each
4518 // assignment would write the value the field already holds, and the
4519 // state a fast→general dot boundary hands on is the same either way.
4520 //
4521 // Until v2.7.6 this block re-wrote them to `true`. Stated as the
4522 // invariant instead, so a future change to the guard that stops
4523 // guaranteeing it fails here in every debug and test build rather than
4524 // having the stores silently paper over it. Measured on its own in
4525 // v2.7.6 (core audit IMP-06, `docs/performance.md`): deleting the
4526 // stores is byte-identical and moves nothing, so this is a statement of
4527 // the invariant, not an optimisation.
4528 debug_assert!(
4529 self.prev_rendering_enabled
4530 && self.rendering_enabled_delayed
4531 && self.rendering_enabled_delayed2,
4532 "fast render path entered without a stably-enabled rendering history"
4533 );
4534
4535 // Sprite-evaluation FSM (visible scanline) + isolated OAM data-bus model.
4536 self.tick_sprite_eval_per_dot();
4537 self.tick_oam_bus();
4538
4539 // Background fetch pipeline: dots 1..=256 are all in the fetch window,
4540 // `phase = (dot - 1) & 7`.
4541 let phase = dot.wrapping_sub(1) & 7;
4542 // Phase 0: shift-register reload (reload gate == `bg_reload_render`).
4543 if phase == 0 {
4544 self.reload_bg_shift_regs();
4545 }
4546 // 2-cycle-ALE address-latch half (even phases).
4547 match phase {
4548 0 => self.ale_drive_nt(),
4549 2 => self.ale_drive_at(),
4550 4 => self.ale_drive_bg_lo(),
4551 6 => self.ale_drive_bg_hi(),
4552 _ => {}
4553 }
4554 // Read half (odd phases).
4555 match phase {
4556 1 => self.fetch_nt(bus),
4557 3 => self.fetch_at(bus),
4558 5 => self.fetch_bg_lo(bus),
4559 7 => self.fetch_bg_hi(bus),
4560 _ => {}
4561 }
4562 self.observe_bg_a12_lead(bus, phase);
4563 // Phase 7 (cycle 8 of the group): coarse-X increment. The dots
4564 // 321..=336 prefetch `<<= 8` is out of the 1..=256 range, so it never
4565 // applies here.
4566 if phase == 7 {
4567 self.inc_hori_v();
4568 }
4569 // Dot 256: vertical-V increment (with the 29→0 wrap-and-flip quirk).
4570 if dot == 256 {
4571 self.inc_vert_v();
4572 }
4573
4574 // Pixel emission + BG shift. The shift gate `render_line && rendering`
4575 // is `true` throughout the covered window.
4576 self.emit_pixel();
4577 self.shift_bg();
4578 }
4579
4580 // ------------------------------------------------------------------
4581 // Background fetch + shift + increment helpers.
4582 // ------------------------------------------------------------------
4583
4584 /// Fetch the nametable byte for the current `v`. Address: `$2000 |
4585 /// (v & 0x0FFF)`.
4586 ///
4587 /// MMC5 vertical split-screen: at the boundary of each 8-dot BG fetch
4588 /// group, the mapper is consulted via `bus.bg_split_state(...)`. If
4589 /// the current tile column falls within the alt region, the returned
4590 /// state supplies the synthesized NT / AT addresses, the alt fine-Y,
4591 /// and the 4 KiB CHR bank index. We latch it onto `bg_split_latch` for
4592 /// consumption by AT / BG-lo / BG-hi within the same fetch group.
4593 #[allow(clippy::cast_sign_loss)]
4594 #[inline]
4595 fn fetch_nt<B: PpuBus>(&mut self, bus: &mut B) {
4596 // Compute the (scanline_y, coarse_x) the alt region would be sampled
4597 // at. The pre-render line passes 0 (the alt region only renders on
4598 // visible lines, but the query is benign for pre-render).
4599 let scanline_y = if self.scanline < 0 {
4600 0
4601 } else {
4602 self.scanline as u16
4603 };
4604 let coarse_x = self.v & 0x001F;
4605 // NOTE (v2.0.3 / ADR 0030, Option 1): the MMC5 vertical-split query stays
4606 // HERE at the read dot — its `split_chr_bank_latch` side effect (which
4607 // `nametable_fetch`/`chr_offset` read) is mapper-observable, and moving it
4608 // one dot earlier to a phase-0 ALE shifts the Uchuu Keibitai SDF split
4609 // rendering. Consequently the NT fetch's octal-latch load co-locates with
4610 // its read (via `ale_splice`'s not-armed path below) rather than a phase-0
4611 // ALE; the AT / pattern fetches ARE true two-dot ALEs. See the plan.
4612 self.bg_split_latch = bus.bg_split_state(scanline_y, coarse_x);
4613
4614 let nt_addr = if let Some(split) = self.bg_split_latch {
4615 split.nt_addr
4616 } else {
4617 0x2000 | (self.v & 0x0FFF)
4618 };
4619 // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half. For the common
4620 // (non-split) case the phase-0 NT ALE already drove the plain address and
4621 // loaded `octal_latch`, so `ale_splice` takes its armed path and the read
4622 // address is the true multiplexed splice `(address_bus & 0x3F00) |
4623 // octal_latch` — transparent for a coherent fetch, and the divergence
4624 // point for the delayed-`CopyV` "Hybrid Addresses" corruption. For an
4625 // MMC5-split fetch the phase-0 ALE drove the PLAIN address (the split
4626 // query lives HERE for its mapper-observable side effect), so disarm and
4627 // read `split.nt_addr` co-located instead — byte-identical to a coherent
4628 // fetch.
4629 let nt_addr = {
4630 if self.bg_split_latch.is_some() {
4631 self.ale_armed = false;
4632 }
4633 self.ale_splice(nt_addr)
4634 };
4635 self.nt_latch = self.read_vram(bus, nt_addr);
4636 // v2.3.2 "Lucid" — capture the address this tile's number came from. The
4637 // SPLICED address, i.e. the one actually driven, so a hybrid-address
4638 // corruption shows the address the hardware really read rather than the
4639 // one it meant to. Telemetry only.
4640 #[cfg(feature = "debug-hooks")]
4641 {
4642 self.prov_nt_pending = nt_addr;
4643 }
4644 // Data phase: drive the byte just read back onto the multiplexed bus's low
4645 // 8 bits (AD7-0). Behavior-neutral (the next fetch's ALE overwrites it).
4646 self.ale_drive_data(self.nt_latch);
4647 // Latch any per-tile extended-attribute info (MMC5 ExGrafix). Skip
4648 // when split is active: the alt region uses standard 4-bit AT
4649 // semantics, not ExGrafix.
4650 self.ex_attr_latch = if self.bg_split_latch.is_some() {
4651 None
4652 } else {
4653 bus.peek_ex_attribute(self.v)
4654 };
4655 }
4656
4657 /// Fetch the attribute byte for the current `v`. Address:
4658 /// `$23C0 | (v & 0x0C00) | ((v >> 4) & 0x38) | ((v >> 2) & 0x07)`.
4659 #[inline]
4660 fn fetch_at<B: PpuBus>(&mut self, bus: &mut B) {
4661 // Split active: use the alt AT address and recover coarse-X / coarse-Y
4662 // from the latched split state's NT address (where coarse-X = bits
4663 // 0..=4, coarse-Y = bits 5..=9).
4664 if let Some(split) = self.bg_split_latch {
4665 let at_addr = split.at_addr;
4666 // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half (split AT path):
4667 // splice / consume the ALE arm so it can't leak to the next fetch.
4668 let at_addr = self.ale_splice(at_addr);
4669 let byte = self.read_vram(bus, at_addr);
4670 // v2.3.2 "Lucid" — the split's own attribute address, which the
4671 // standard `$23C0 | ...` arithmetic cannot reproduce. This branch is
4672 // exactly why the record carries `at` instead of deriving it.
4673 #[cfg(feature = "debug-hooks")]
4674 {
4675 self.prov_at_pending = at_addr;
4676 }
4677 self.ale_drive_data(byte);
4678 let coarse_x = (split.nt_addr & 0x001F) as u8;
4679 let coarse_y = ((split.nt_addr >> 5) & 0x001F) as u8;
4680 let shift = ((coarse_y & 0x02) << 1) | (coarse_x & 0x02);
4681 self.at_latch = (byte >> shift) & 0x03;
4682 return;
4683 }
4684 let v = self.v;
4685 let at_addr = 0x23C0 | (v & 0x0C00) | ((v >> 4) & 0x38) | ((v >> 2) & 0x07);
4686 // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half (normal AT path).
4687 let at_addr = self.ale_splice(at_addr);
4688 let byte = self.read_vram(bus, at_addr);
4689 // v2.3.2 "Lucid" — see the split branch above.
4690 #[cfg(feature = "debug-hooks")]
4691 {
4692 self.prov_at_pending = at_addr;
4693 }
4694 self.ale_drive_data(byte);
4695 // Pick the 2-bit attribute based on coarse-X[1] and coarse-Y[1].
4696 let coarse_x = (v & 0x1F) as u8;
4697 let coarse_y = ((v >> 5) & 0x1F) as u8;
4698 let shift = ((coarse_y & 0x02) << 1) | (coarse_x & 0x02);
4699 let standard_palette = (byte >> shift) & 0x03;
4700 // ExGrafix override: replace the 2-bit palette with the per-tile
4701 // value latched at NT-fetch time.
4702 self.at_latch = self
4703 .ex_attr_latch
4704 .map_or(standard_palette, |ex| ex.palette & 0x03);
4705 }
4706
4707 /// Fetch BG pattern low byte for the current `nt_latch` + fine-Y of `v`.
4708 ///
4709 /// In MMC5 `ExGrafix` mode the mapper has internally latched a per-tile
4710 /// 4 KiB CHR bank from the most recent `peek_ex_attribute` call; it
4711 /// will resolve this `addr` against that bank rather than the standard
4712 /// BG bank registers. No address-bus rerouting required.
4713 ///
4714 /// In MMC5 vertical split-screen mode the mapper has likewise latched
4715 /// the `$5202` 4 KiB CHR bank from the most recent `bg_split_state`
4716 /// call, and the alt fine-Y replaces `v`'s fine-Y.
4717 #[inline]
4718 fn fetch_bg_lo<B: PpuBus>(&mut self, bus: &mut B) {
4719 let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4720 let fine_y = self
4721 .bg_split_latch
4722 .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4723 let addr = bg_table | (u16::from(self.nt_latch) << 4) | fine_y;
4724 self.observe_a12_addr(bus, addr);
4725 // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half: A12 above stays on the
4726 // INTENDED `addr`; only the DATA read address goes through the ALE splice
4727 // (stale-latch "ALE + Read"). `addr` itself is preserved for the hd-pack
4728 // tile-base latch below.
4729 let read_addr = self.ale_splice(addr);
4730 self.bg_lo_latch = self.read_vram(bus, read_addr);
4731 // v2.3.2 "Lucid" — the pattern ROW address (fine-Y kept, unlike the
4732 // `hd-pack` latch below which masks it off to get the 16-byte tile base):
4733 // provenance answers "which CHR byte fed THIS pixel", which is a row, not
4734 // a tile. The SPLICED address again, so a hybrid-address corruption shows
4735 // what was really read.
4736 #[cfg(feature = "debug-hooks")]
4737 {
4738 self.prov_bg_latch = ProvBgAddrs {
4739 nt: self.prov_nt_pending,
4740 at: self.prov_at_pending,
4741 pattern: read_addr,
4742 };
4743 }
4744 self.ale_drive_data(self.bg_lo_latch);
4745 // v1.2.0 C3 (hd-pack): latch the 16-byte tile base (fine-Y masked off)
4746 // for this fetch group. Promoted into the `hd_bg_addr_*` queue at the
4747 // next shifter reload so it tracks the BG pattern shifters tile-for-tile.
4748 // Output-only; no new VRAM read, no A12.
4749 #[cfg(feature = "hd-pack")]
4750 {
4751 self.hd_bg_addr_latch = addr & 0x1FF0;
4752 // CHR-ROM absolute tile index (offset/16), or the CHR-RAM sentinel.
4753 self.hd_bg_idx_latch = bus.chr_phys(addr).map_or(HD_CHR_RAM, |o| o / 16);
4754 }
4755 }
4756
4757 /// Fetch BG pattern high byte (offset +8 from the low fetch).
4758 #[inline]
4759 fn fetch_bg_hi<B: PpuBus>(&mut self, bus: &mut B) {
4760 let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4761 let fine_y = self
4762 .bg_split_latch
4763 .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4764 let addr = bg_table | (u16::from(self.nt_latch) << 4) | 0x08 | fine_y;
4765 self.observe_a12_addr(bus, addr);
4766 // v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE read half (see `fetch_bg_lo`):
4767 // A12 above stays on the INTENDED `addr`; only the DATA read address goes
4768 // through the ALE splice (stale-latch "ALE + Read").
4769 let read_addr = self.ale_splice(addr);
4770 self.bg_hi_latch = self.read_vram(bus, read_addr);
4771 self.ale_drive_data(self.bg_hi_latch);
4772 }
4773
4774 // === v2.0.3 (ADR 0030, Option 1) — 2-cycle-ALE fetch model ===============
4775 //
4776 // A genuine two-dot VRAM transaction. The attribute + pattern fetches' EVEN
4777 // dot (the ALE half, phases 2/4/6) drives the full 14-bit address onto
4778 // `address_bus` and captures its low byte into `octal_latch` via
4779 // [`Self::drive_bus`]; the following ODD dot (the read half, phases 3/5/7 —
4780 // the existing `fetch_*`) resolves the effective address through
4781 // [`Self::ale_splice`] and drives the DATA byte back onto the low bus via
4782 // [`Self::ale_drive_data`]. For a coherent fetch the address the ALE drove
4783 // equals the address the read would compute (`v` is constant across the 8-dot
4784 // group's phases 0..=6; the coarse-X increment is at phase 7 AFTER the read),
4785 // so the splice returns the intended address and this is behavior-neutral.
4786 //
4787 // The NAMETABLE fetch is the exception: its MMC5 vertical-split query
4788 // (`bg_split_state`, whose `split_chr_bank_latch` side effect
4789 // `nametable_fetch`/`chr_offset` read) is mapper-observable and must fire at
4790 // the read dot (phase 1), so the NT octal-latch load co-locates with the read
4791 // via `ale_splice`'s not-armed path (there is no phase-0 NT ALE); the NT ALE
4792 // (`ale_drive_nt`) still drives the plain address so the latch naturally
4793 // carries the one-tile-ahead NT-low the "Hybrid Addresses" corruption needs.
4794
4795 /// ALE half of the nametable fetch (phase 0). Drives the PLAIN nametable
4796 /// address `0x2000 | (v & 0x0FFF)` and loads the octal latch with its low
4797 /// byte — a true two-dot ALE for the common (non-split) case, which is what
4798 /// lets `octal_latch` naturally carry the one-tile-ahead NT-low the "Hybrid
4799 /// Addresses" corruption needs. The MMC5-split query is deferred to the read
4800 /// dot (phase 1, `fetch_nt`); when it turns out split-active, `fetch_nt`
4801 /// disarms this ALE and reads the synthesized `split.nt_addr` co-located.
4802 const fn ale_drive_nt(&mut self) {
4803 let nt_addr = 0x2000 | (self.v & 0x0FFF);
4804 self.drive_bus(nt_addr, false);
4805 }
4806
4807 /// ALE half of the attribute fetch (phase 2). Uses the `bg_split_latch`
4808 /// already set by the nametable read at phase 1.
4809 const fn ale_drive_at(&mut self) {
4810 let at_addr = if let Some(split) = self.bg_split_latch {
4811 split.at_addr
4812 } else {
4813 let v = self.v;
4814 0x23C0 | (v & 0x0C00) | ((v >> 4) & 0x38) | ((v >> 2) & 0x07)
4815 };
4816 self.drive_bus(at_addr, false);
4817 }
4818
4819 /// ALE half of the BG pattern-low fetch (phase 4). Uses `nt_latch` (set at
4820 /// the phase-1 nametable read) and `v`'s fine-Y (or the split fine-Y).
4821 fn ale_drive_bg_lo(&mut self) {
4822 let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4823 let fine_y = self
4824 .bg_split_latch
4825 .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4826 let addr = bg_table | (u16::from(self.nt_latch) << 4) | fine_y;
4827 self.drive_bus(addr, true);
4828 }
4829
4830 /// ALE half of the BG pattern-high fetch (phase 6). Same as the low plane
4831 /// with bit 3 set (the +8 byte offset).
4832 fn ale_drive_bg_hi(&mut self) {
4833 let bg_table = u16::from(self.ctrl.contains(PpuCtrl::BG_PATTERN_HIGH)) << 12;
4834 let fine_y = self
4835 .bg_split_latch
4836 .map_or((self.v >> 12) & 0x07, |s| u16::from(s.fine_y) & 0x07);
4837 let addr = bg_table | (u16::from(self.nt_latch) << 4) | 0x08 | fine_y;
4838 self.drive_bus(addr, true);
4839 }
4840
4841 /// Drive a full 14-bit fetch address onto the multiplexed bus (the ALE
4842 /// half): `address_bus` takes the whole address and the 74LS373 octal latch
4843 /// captures A7-A0. Arms `ale_armed` for the matching read half.
4844 ///
4845 /// `is_pattern` marks the two BG-pattern ALEs (phases 4/6). While the "ALE +
4846 /// Read" freeze (`pattern_latch_stale`) is pending — a `$2007`-read ALE
4847 /// overlapped the fetch cadence and froze `octal_latch` on the read's DATA
4848 /// byte — the latch is NOT reloaded (the frozen DATA byte survives across any
4849 /// intervening ALE); the first pattern ALE afterwards consumes the flag, so
4850 /// its read splices `(PAR high 6):(stale DATA low 8)` = `$0FFF`.
4851 const fn drive_bus(&mut self, addr: u16, is_pattern: bool) {
4852 self.address_bus = addr;
4853 if self.pattern_latch_stale {
4854 if is_pattern {
4855 self.pattern_latch_stale = false;
4856 }
4857 } else {
4858 self.octal_latch = (addr & 0xFF) as u8;
4859 }
4860 self.ale_armed = true;
4861 }
4862
4863 /// Resolve a fetch's effective read address through the multiplexed bus (the
4864 /// read half). When a real ALE preceded this read (`ale_armed`), the address
4865 /// is the splice of the ALE-driven high 6 bits with the latched low 8:
4866 /// `(address_bus & 0x3F00) | octal_latch` — transparent for a coherent fetch
4867 /// (Phase 1), the divergence point for the Phase-3 corruptions. With NO
4868 /// preceding ALE (the dot-337-340 garbage nametable fetches), drive + latch
4869 /// `intended` in place so the read stays behavior-neutral.
4870 #[allow(clippy::missing_const_for_fn)] // u16::from is not yet const-stable
4871 fn ale_splice(&mut self, intended: u16) -> u16 {
4872 if self.ale_armed {
4873 self.ale_armed = false;
4874 // v2.3.6 — high 6 bits from `intended` (recomputed from the LIVE `v` at
4875 // the read dot), not from the ALE-time `address_bus` snapshot. Upstream
4876 // AccuracyCoin's commentary was rewritten to say the address bus is
4877 // driven EVERY ppu cycle and its upper 6 bits track `v`, so the hybrid
4878 // address is what a continuously-driven bus produces when `v` moves
4879 // between a fetch's ALE half and its read half. Behaviour-neutral for a
4880 // coherent fetch: `v` unchanged => `intended` == the driven address.
4881 let effective = (intended & 0x3F00) | u16::from(self.octal_latch);
4882 // Diagnostic: record any read whose spliced effective address diverges
4883 // from the intended one (the two corruptions) for the TriCNES per-dot
4884 // cross-diff. `push` self-filters to scanlines 2-5, so this is cheap.
4885 if effective != intended {
4886 octal_trace::push(
4887 if intended >= 0x2000 {
4888 octal_trace::K_HYBRID
4889 } else {
4890 octal_trace::K_STALE
4891 },
4892 self.frame,
4893 self.scanline,
4894 self.dot,
4895 u32::from(effective),
4896 );
4897 }
4898 effective
4899 } else {
4900 self.address_bus = intended;
4901 self.octal_latch = (intended & 0xFF) as u8;
4902 intended
4903 }
4904 }
4905
4906 /// Data half of a VRAM access: drive the byte just read back onto the
4907 /// multiplexed bus's low 8 bits (AD7-0). `octal_latch` is NOT refreshed here
4908 /// (the 74LS373 latch holds the ADDRESS low from the ALE) — that retention is
4909 /// what a `$2007`-read ALE overlap exploits (the "ALE + Read" corruption). It
4910 /// is otherwise transparent: the next fetch's ALE overwrites `address_bus`
4911 /// wholesale.
4912 #[allow(clippy::missing_const_for_fn)] // u16::from is not yet const-stable
4913 fn ale_drive_data(&mut self, data: u8) {
4914 self.address_bus = (self.address_bus & 0xFF00) | u16::from(data);
4915 }
4916
4917 /// Shift the BG pattern and attribute shift registers by one bit.
4918 ///
4919 /// All four registers are 16-bit and advance in lockstep so the
4920 /// attribute palette tracks the same tile column as the pattern bits.
4921 const fn shift_bg(&mut self) {
4922 self.bg_shift_lo <<= 1;
4923 self.bg_shift_hi <<= 1;
4924 self.at_shift_lo <<= 1;
4925 self.at_shift_hi <<= 1;
4926 // v2.0 Phase 6 (mc-ppu-subpos): BG-shifter SERIAL-IN. Per nesdev "PPU
4927 // signals", the bit shifted into the pattern shifters from the right is
4928 // a constant 0 for the LOW plane and a constant 1 for the HIGH plane.
4929 // It is normally invisible: the dot%8==1 reload overwrites bits 0-7
4930 // every 8 shifts, so the injected '1' never reaches the output bits
4931 // 8-15 before being washed (=> framebuffer byte-identical for normal
4932 // rendering, oracle-safe). It surfaces ONLY when a precisely-timed
4933 // `$2001` render-toggle SKIPS a reload while shifting continues, drawing
4934 // opaque BG pixels on an all-translucent nametable (AccuracyCoin "BG
4935 // Serial In"). The attribute shifters have no serial-in (the test only
4936 // needs a non-transparent pattern bit, not a specific palette).
4937 {
4938 self.bg_shift_hi |= 1;
4939 }
4940 }
4941
4942 /// Pre-fetch (dots 328 / 336) byte shift: advance all four BG shift
4943 /// registers by 8 bits in lockstep, moving the just-reloaded tile
4944 /// data from bits 0-7 to bits 8-15 and clearing bits 0-7 for the next
4945 /// reload. This substitutes for the per-cycle `shift_bg` that does not
4946 /// run during the dots 321-336 pre-fetch region. The attribute
4947 /// registers MUST shift identically to the pattern registers here —
4948 /// omitting them was the 086ce4d left-edge palette regression.
4949 #[inline]
4950 const fn prefetch_shift_bg_regs(&mut self) {
4951 self.bg_shift_lo <<= 8;
4952 self.bg_shift_hi <<= 8;
4953 self.at_shift_lo <<= 8;
4954 self.at_shift_hi <<= 8;
4955 // v1.2.0 C3 (hd-pack): the `<<= 8` promotes the low (next) tile into the
4956 // high (displayed) byte — mirror the address queue. Telemetry only.
4957 #[cfg(feature = "hd-pack")]
4958 {
4959 self.hd_bg_addr_cur = self.hd_bg_addr_next;
4960 self.hd_bg_idx_cur = self.hd_bg_idx_next;
4961 }
4962 // v2.3.2 "Lucid": same promotion for the provenance cascade.
4963 //
4964 // A/B'd rather than assumed. For the VISIBLE region this is redundant —
4965 // every displayed tile passes through a `reload_bg_shift_regs` that
4966 // overwrites `cur` from `next` anyway, and the provenance test passes
4967 // identically with this removed. It is kept because this function is also
4968 // called on the rendering-DISABLE edge (dots 329-336) to complete a
4969 // frozen group's pending shift, and there pixels can be emitted from the
4970 // shifters before any further reload — so mirroring the shift is what
4971 // keeps the reported tile matching the one actually on screen.
4972 #[cfg(feature = "debug-hooks")]
4973 {
4974 self.prov_bg_cur = self.prov_bg_next;
4975 }
4976 }
4977
4978 /// Reload the low bytes of the BG pattern and attribute shift
4979 /// registers from the latched fetch bytes.
4980 ///
4981 /// The 2-bit attribute is constant across all 8 pixels of a tile, so
4982 /// each attribute bit is expanded to a full `0xFF`/`0x00` byte into
4983 /// bits 0-7 — the same low-byte slot the pattern bytes occupy. This
4984 /// keeps the attribute shifter bit-for-bit aligned with the pattern
4985 /// shifters through both the per-cycle shifts (dots 1-256) and the
4986 /// pre-fetch `<<= 8` (dots 328 / 336).
4987 #[inline]
4988 const fn reload_bg_shift_regs(&mut self) {
4989 self.bg_shift_lo = (self.bg_shift_lo & 0xFF00) | self.bg_lo_latch as u16;
4990 self.bg_shift_hi = (self.bg_shift_hi & 0xFF00) | self.bg_hi_latch as u16;
4991 let at_lo = if (self.at_latch & 0x01) != 0 {
4992 0xFF
4993 } else {
4994 0x00
4995 };
4996 let at_hi = if (self.at_latch & 0x02) != 0 {
4997 0xFF
4998 } else {
4999 0x00
5000 };
5001 self.at_shift_lo = (self.at_shift_lo & 0xFF00) | at_lo;
5002 self.at_shift_hi = (self.at_shift_hi & 0xFF00) | at_hi;
5003 // v1.2.0 C3 (hd-pack): mirror the pattern reload — the prior `next`
5004 // tile is now in the high byte (displayed), and the freshly-latched
5005 // tile fills the low byte. Telemetry only; no state effect.
5006 #[cfg(feature = "hd-pack")]
5007 {
5008 self.hd_bg_addr_cur = self.hd_bg_addr_next;
5009 self.hd_bg_addr_next = self.hd_bg_addr_latch;
5010 self.hd_bg_idx_cur = self.hd_bg_idx_next;
5011 self.hd_bg_idx_next = self.hd_bg_idx_latch;
5012 }
5013 // v2.3.2 "Lucid": same promotion for the provenance address cascade —
5014 // one struct copy per stage, so the three addresses cannot drift apart.
5015 #[cfg(feature = "debug-hooks")]
5016 {
5017 self.prov_bg_cur = self.prov_bg_next;
5018 self.prov_bg_next = self.prov_bg_latch;
5019 }
5020 }
5021
5022 /// Increment coarse X with nametable-X wrap.
5023 ///
5024 /// Note: this is an internal loopy-register increment. It does NOT
5025 /// drive the PPU address bus, so it must not emit A12 transitions —
5026 /// the address bus stays on the last-fetched address (BG-high) until
5027 /// the next fetch. An earlier version of this code called
5028 /// `observe_a12` here, which spuriously interpreted `v`'s fine-Y bit
5029 /// 0 as A12 and produced ~16 false A12 rising edges per scanline,
5030 /// breaking MMC3's IRQ count (which expects exactly 1 rise per
5031 /// rendered scanline, at PPU dot ~260, with standard pattern-table
5032 /// layout).
5033 const fn inc_hori_v(&mut self) {
5034 if (self.v & 0x001F) == 31 {
5035 self.v &= !0x001F;
5036 self.v ^= 0x0400;
5037 } else {
5038 self.v += 1;
5039 }
5040 }
5041
5042 /// Increment fine Y, with the 29->0 wrap-and-flip-nametable-Y quirk.
5043 ///
5044 /// Same A12 caveat as [`Self::inc_hori_v`]: this is an internal
5045 /// register increment, not an address-bus driver.
5046 const fn inc_vert_v(&mut self) {
5047 if (self.v & 0x7000) == 0x7000 {
5048 self.v &= !0x7000;
5049 let mut y = (self.v & 0x03E0) >> 5;
5050 if y == 29 {
5051 y = 0;
5052 self.v ^= 0x0800;
5053 } else if y == 31 {
5054 y = 0;
5055 } else {
5056 y += 1;
5057 }
5058 self.v = (self.v & !0x03E0) | (y << 5);
5059 } else {
5060 self.v += 0x1000;
5061 }
5062 }
5063
5064 /// Copy horizontal bits of `t` into `v` (bits 0-4 + 10).
5065 const fn copy_hori_t_to_v(&mut self) {
5066 self.v = (self.v & !0x041F) | (self.t & 0x041F);
5067 }
5068
5069 /// Copy vertical bits of `t` into `v` (bits 5-9 + 11-14).
5070 const fn copy_vert_t_to_v(&mut self) {
5071 self.v = (self.v & !0x7BE0) | (self.t & 0x7BE0);
5072 }
5073
5074 // ------------------------------------------------------------------
5075 // Pixel emission.
5076 // ------------------------------------------------------------------
5077
5078 /// Emit one pixel into the framebuffer at the current `(scanline, dot)`.
5079 #[allow(clippy::cast_sign_loss)]
5080 #[allow(clippy::too_many_lines)] // + the ppu-sprite-shifter-counter X-counter/shift loop
5081 fn emit_pixel(&mut self) {
5082 let pixel_x = self.dot - 1;
5083 let pixel_y = self.scanline as u16; // already validated >= 0 by caller
5084 let fx = self.x;
5085 // BG pixel (bits 0-1 = pattern, bits 2-3 = palette)
5086 let (bg_idx, bg_pal) = if self.mask.contains(PpuMask::SHOW_BG)
5087 && (pixel_x >= 8 || self.mask.contains(PpuMask::SHOW_BG_LEFT))
5088 {
5089 let mask = 0x8000u16 >> fx;
5090 let p0 = u8::from((self.bg_shift_lo & mask) != 0);
5091 let p1 = u8::from((self.bg_shift_hi & mask) != 0);
5092 let idx = (p1 << 1) | p0;
5093 let a0 = u8::from((self.at_shift_lo & mask) != 0);
5094 let a1 = u8::from((self.at_shift_hi & mask) != 0);
5095 (idx, (a1 << 1) | a0)
5096 } else {
5097 (0, 0)
5098 };
5099
5100 // Sprite pixel evaluation (Sprint 2-3).
5101 let mut spr_idx: u8 = 0;
5102 let mut spr_pal: u8 = 0;
5103 let mut spr_priority_front = false;
5104 let mut spr_zero_pixel = false;
5105 #[cfg(any(feature = "hd-pack", feature = "debug-hooks"))]
5106 let mut spr_slot: usize = 0;
5107 // v1.8.9 — every opaque sprite covering this pixel (slot indices), for the
5108 // HD-pack multi-sprite conditions. Collected but never consulted by the
5109 // winner logic below, so the framebuffer stays byte-identical.
5110 #[cfg(feature = "hd-pack")]
5111 let mut hd_sprites: [usize; 4] = [0; 4];
5112 #[cfg(feature = "hd-pack")]
5113 let mut hd_spr_n: usize = 0;
5114 if self.mask.contains(PpuMask::SHOW_SPRITE)
5115 && (pixel_x >= 8 || self.mask.contains(PpuMask::SHOW_SPRITE_LEFT))
5116 {
5117 for i in 0..self.spr_count as usize {
5118 // v2.0 (ppu-sprite-shifter-counter): a sprite emits when its
5119 // X-counter is 0 OR it is in the persistent halted state. The
5120 // `spr_x == 0` term keeps the legacy px-0 emit timing (an X=0
5121 // sprite re-armed at dot 339 must emit at px 0, not px 1 — else a
5122 // spurious Sprite-0-Hit test-8 hit), while `spr_halted` carries
5123 // Stale Sprite t5/6. Default build: the legacy `spr_x == 0`.
5124 let emit_active = self.spr_x[i] == 0 || self.spr_halted[i];
5125 if !emit_active {
5126 continue;
5127 }
5128 let lo = u8::from((self.spr_shift_lo[i] & 0x80) != 0);
5129 let hi = u8::from((self.spr_shift_hi[i] & 0x80) != 0);
5130 let val = (hi << 1) | lo;
5131 if val == 0 {
5132 continue;
5133 }
5134 // hd-pack: record every opaque sprite covering this pixel.
5135 #[cfg(feature = "hd-pack")]
5136 if hd_spr_n < 4 {
5137 hd_sprites[hd_spr_n] = i;
5138 hd_spr_n += 1;
5139 }
5140 // The first opaque sprite (priority order) is the VISIBLE winner;
5141 // `spr_idx == 0` gates it so only the first sets the render state.
5142 if spr_idx == 0 {
5143 spr_idx = val;
5144 spr_pal = self.spr_attr[i] & 0x03;
5145 spr_priority_front = (self.spr_attr[i] & 0x20) == 0;
5146 #[cfg(any(feature = "hd-pack", feature = "debug-hooks"))]
5147 {
5148 spr_slot = i;
5149 }
5150 if i == 0 && self.spr_zero_in_line {
5151 spr_zero_pixel = true;
5152 }
5153 }
5154 // Default build stops at the winner (byte-identical); the hd-pack
5155 // build keeps scanning to collect the hidden sprites above (which
5156 // never touch the winner state, so the framebuffer is unchanged).
5157 #[cfg(not(feature = "hd-pack"))]
5158 break;
5159 }
5160 }
5161 // v3.1.0 (`T-SPRITE-LIMIT`): the sprites beyond the eighth, only where
5162 // none of the eight hardware sprites is opaque (they have the higher
5163 // OAM indexes, so the lower priority). Never sprite 0, so never a hit.
5164 // `spr_extra_count` is 0 unless the option is on.
5165 if spr_idx == 0
5166 && self.spr_extra_count != 0
5167 && self.mask.contains(PpuMask::SHOW_SPRITE)
5168 && (pixel_x >= 8 || self.mask.contains(PpuMask::SHOW_SPRITE_LEFT))
5169 {
5170 for e in 0..usize::from(self.spr_extra_count) {
5171 let off = pixel_x.wrapping_sub(u16::from(self.spr_extra_x[e]));
5172 if off >= 8 {
5173 continue;
5174 }
5175 let bit = 7 - off;
5176 let lo = (self.spr_extra_lo[e] >> bit) & 1;
5177 let hi = (self.spr_extra_hi[e] >> bit) & 1;
5178 let val = (hi << 1) | lo;
5179 if val != 0 {
5180 spr_idx = val;
5181 spr_pal = self.spr_extra_attr[e] & 0x03;
5182 spr_priority_front = (self.spr_extra_attr[e] & 0x20) == 0;
5183 break;
5184 }
5185 }
5186 }
5187
5188 // Combine BG + sprite per priority.
5189 //
5190 // v2.3.2 "Lucid": the priority chain now yields the palette ADDRESS and
5191 // the single `read_palette` happens after it, instead of each arm reading
5192 // inline. Semantically identical, and it makes the address available to
5193 // the provenance record below — so the panel reports the exact `$3Fxx`
5194 // this pixel came from, including the `$3F10` family pre-mirroring and
5195 // the rendering-disabled backdrop-override address, rather than
5196 // re-deriving it from the priority result and getting the corners wrong.
5197 let pal_addr: u16 = if bg_idx == 0 && spr_idx == 0 {
5198 // Universal background ($3F00) — EXCEPT the palette backdrop-override
5199 // (F1.1): with rendering DISABLED and the VRAM address `v` pointing
5200 // into palette space ($3F00-$3FFF), the palette's shared address line
5201 // is driven by `v`, so hardware outputs the color at `v & 0x1F`
5202 // INSTEAD of the backdrop (`NESdev` "PPU palettes"; Mesen2 `NesPpu.cpp`
5203 // / ares output stage). This is a DISPLAY artifact only — palette RAM
5204 // is not mutated. It cannot fire while rendering is enabled: there
5205 // the fetch pipeline owns `v` and this branch means a transparent
5206 // pixel, which is the genuine backdrop. `read_palette` applies the
5207 // $10/$14/$18/$1C mirror + greyscale, so the override is mirror- and
5208 // greyscale-correct with no extra handling.
5209 if !self.mask.rendering_enabled() && (self.v & 0x3F00) == 0x3F00 {
5210 0x3F00 | (self.v & 0x1F)
5211 } else {
5212 0x3F00
5213 }
5214 } else if bg_idx == 0 {
5215 0x3F10 | (u16::from(spr_pal) << 2) | u16::from(spr_idx)
5216 } else if spr_idx == 0 {
5217 0x3F00 | (u16::from(bg_pal) << 2) | u16::from(bg_idx)
5218 } else {
5219 // Both opaque. Sprite-0 hit detection (constraints per nesdev).
5220 if spr_zero_pixel
5221 && pixel_x < 255
5222 && !(pixel_x < 8
5223 && (!self.mask.contains(PpuMask::SHOW_BG_LEFT)
5224 || !self.mask.contains(PpuMask::SHOW_SPRITE_LEFT)))
5225 {
5226 self.status.insert(PpuStatus::SPRITE_ZERO_HIT);
5227 }
5228 if spr_priority_front {
5229 0x3F10 | (u16::from(spr_pal) << 2) | u16::from(spr_idx)
5230 } else {
5231 0x3F00 | (u16::from(bg_pal) << 2) | u16::from(bg_idx)
5232 }
5233 };
5234 // One read at the end instead of one per arm. `read_palette` is a pure
5235 // read (the greyscale mask it consults is not touched by the sprite-0-hit
5236 // insert above), so hoisting it out of the branches is behaviour-
5237 // preserving as well as what clippy's `branches_sharing_code` wants.
5238 let final_idx = self.read_palette(pal_addr) & 0x3F;
5239
5240 // Write RGBA8 to framebuffer.
5241 let off = ((pixel_y as usize) * 256 + pixel_x as usize) * 4;
5242 // v2.8.0 Phase 4 — route through the precomputed
5243 // `(emphasis << 6) | color` lookup (built from the same pure
5244 // `palette_color_to_rgba`, so byte-identical to the old per-pixel
5245 // call for both the 2C02 composite default and the Vs./PC10 RGB
5246 // palettes) and store all four bytes with one bounds-checked slice
5247 // copy instead of four indexed stores.
5248 // v3.1.0 (`T-PAL-EMPHASIS`): the index is the PHYSICAL tint (bit 0
5249 // red, bit 1 green, bit 2 blue). PPUMASK bit 5 is red on the NTSC 2C02
5250 // and GREEN on the PAL 2C07 and the Dendy, bit 6 the reverse (NESdev
5251 // "Colour emphasis"), so those two exchange off NTSC. The region is
5252 // fixed per console, so the branch is constant.
5253 let raw = (self.mask.bits() >> 5) & 0x07;
5254 let emph = usize::from(if matches!(self.region, PpuRegion::Ntsc) {
5255 raw
5256 } else {
5257 (raw & 0b100) | ((raw & 0b001) << 1) | ((raw & 0b010) >> 1)
5258 });
5259 let lut_idx = (emph << 6) | usize::from(final_idx);
5260 let rgba = self.rgba_lut[lut_idx];
5261 self.framebuffer[off..off + 4].copy_from_slice(&rgba);
5262 // Parallel palette-index output for the `NES_NTSC` composite filter
5263 // (T-110-A1). Same `(emphasis << 6) | colour` value, in index space;
5264 // `off` is the RGBA byte offset, so `off >> 2` is the pixel index.
5265 // NOTE (v2.3.1 G4): making this store conditional on a consumer wanting
5266 // it was measured by deleting it outright — the ceiling any opt-in gate
5267 // could reach — and the ceiling is ZERO on the shipped configuration.
5268 // `perf` attributes ~0.78% to this line, but a line's sample share is not
5269 // its marginal cost: this is a sequential `u16` store the store buffer
5270 // absorbs off the critical path, so removing it frees nothing and the
5271 // samples simply redistribute. Not worth the correctness hazard of
5272 // gating a buffer the NTSC filter, the mobile API, `fast_dotloop_diff`
5273 // and a unit test all read. See `docs/performance.md`.
5274 self.index_framebuffer[off >> 2] = lut_idx as u16;
5275
5276 // v1.2.0 C3 (hd-pack): record the CHR tile that produced this pixel,
5277 // mirroring the BG-vs-sprite priority decision above. Output-only; this
5278 // reads only already-computed local state, so the framebuffer and all
5279 // timing are byte-identical whether the feature is on or off.
5280 #[cfg(feature = "hd-pack")]
5281 {
5282 // A pixel shows the SPRITE iff the sprite pixel is opaque AND
5283 // (the BG pixel is transparent OR the sprite has front priority) —
5284 // the same condition the `final_idx` priority match encodes.
5285 let shows_sprite = spr_idx != 0 && (bg_idx == 0 || spr_priority_front);
5286 // Multi-sprite telemetry: the identity of every opaque sprite covering
5287 // this pixel (front-to-back), for `spriteAtPosition` / `spriteNearby`.
5288 let hd_sprite_list: [HdSprite; 4] = {
5289 let mut arr = [HdSprite::default(); 4];
5290 for (k, slot) in hd_sprites.iter().take(hd_spr_n).enumerate() {
5291 arr[k] = HdSprite {
5292 chr_tile_index: self.hd_spr_idx[*slot],
5293 palette_colors: self.hd_sprite_palette_colors(self.spr_attr[*slot] & 0x03),
5294 };
5295 }
5296 arr
5297 };
5298 let hd_sprite_n = u8::try_from(hd_spr_n).unwrap_or(4);
5299 let rec = if shows_sprite {
5300 let attr = self.spr_attr[spr_slot];
5301 let flip_h = (attr & 0x40) != 0;
5302 // Column within the sprite (screen X minus the sprite's origin X),
5303 // then flip so the captured offset samples the UNFLIPPED
5304 // replacement directly (composite is flip-free).
5305 // v2.9.5: on the odd-frame-deferred pixel (scanline 0, X=0,
5306 // `spr_rearm_deferred` still set) every drawing sprite emits
5307 // its FIRST column there regardless of its X, so the column is
5308 // 0, not `pixel_x - x`. CodeRabbit on #575: with X=10 the old
5309 // arithmetic gave column 6 and sampled the wrong HD texel.
5310 let col = if self.spr_rearm_deferred {
5311 0
5312 } else {
5313 pixel_x.wrapping_sub(u16::from(self.hd_spr_x[spr_slot])) & 7
5314 };
5315 let off_x = if flip_h { 7 - col } else { col };
5316 HdTileSource {
5317 chr_addr: self.hd_spr_addr[spr_slot],
5318 palette: spr_pal,
5319 is_sprite: true,
5320 flip_h,
5321 flip_v: (attr & 0x80) != 0,
5322 palette_colors: self.hd_sprite_palette_colors(spr_pal),
5323 offset_x: u8::try_from(off_x).unwrap_or(0),
5324 offset_y: self.hd_spr_off_y[spr_slot], // already flip-baked at fetch
5325 chr_tile_index: self.hd_spr_idx[spr_slot],
5326 color_mask: self.mask.bits() & 0xE1,
5327 sprites: hd_sprite_list,
5328 sprite_count: hd_sprite_n,
5329 }
5330 } else if bg_idx != 0 {
5331 // Fine-X picks which of the two shifter tiles this pixel shows +
5332 // the column within it (Mesen usePrev / OffsetX); fine-Y is the row.
5333 let pos = u16::from(fx) + (pixel_x & 7);
5334 let chr = if pos < 8 {
5335 self.hd_bg_addr_cur
5336 } else {
5337 self.hd_bg_addr_next
5338 };
5339 HdTileSource {
5340 chr_addr: chr,
5341 palette: bg_pal,
5342 is_sprite: false,
5343 flip_h: false,
5344 flip_v: false,
5345 palette_colors: self.hd_bg_palette_colors(bg_pal),
5346 offset_x: u8::try_from(pos & 7).unwrap_or(0),
5347 offset_y: u8::try_from((self.v >> 12) & 7).unwrap_or(0),
5348 chr_tile_index: if pos < 8 {
5349 self.hd_bg_idx_cur
5350 } else {
5351 self.hd_bg_idx_next
5352 },
5353 color_mask: self.mask.bits() & 0xE1,
5354 sprites: hd_sprite_list,
5355 sprite_count: hd_sprite_n,
5356 }
5357 } else {
5358 // Universal background — no tile to substitute.
5359 HdTileSource {
5360 chr_addr: HD_TILE_NONE,
5361 palette: 0,
5362 is_sprite: false,
5363 flip_h: false,
5364 flip_v: false,
5365 offset_x: 0,
5366 offset_y: 0,
5367 chr_tile_index: HD_CHR_RAM,
5368 palette_colors: 0,
5369 color_mask: 0,
5370 sprites: hd_sprite_list,
5371 sprite_count: hd_sprite_n,
5372 }
5373 };
5374 self.hd_tile_source[off >> 2] = rec;
5375 }
5376
5377 // v2.3.2 "Lucid" phase 2 — the per-pixel causal record.
5378 //
5379 // Guarded on a plain `bool` rather than `prov_frame.is_some()`: this runs
5380 // 61,440 times a frame in one of the two hottest functions in the
5381 // emulator, so the unarmed cost is one predicted branch on an already-hot
5382 // cache line instead of an `Option` discriminant behind a pointer chase.
5383 // Everything recorded is already computed above or carried in the
5384 // fetch-time cascade — no new VRAM reads, no new arithmetic on the
5385 // shipped path — so the framebuffer and all timing are byte-identical
5386 // whether provenance is armed or not.
5387 #[cfg(feature = "debug-hooks")]
5388 if self.prov_armed {
5389 use crate::provenance::{PATTERN_ADDR_NONE, PixelLayer, PixelProvenance};
5390 // The same condition `final_idx` encoded above: a sprite is visible
5391 // iff it is opaque AND (the BG is transparent OR it has front
5392 // priority). Re-deriving it here rather than threading a flag keeps
5393 // the shipped path free of a variable that only telemetry reads.
5394 let shows_sprite = spr_idx != 0 && (bg_idx == 0 || spr_priority_front);
5395 let layer = if shows_sprite {
5396 PixelLayer::Sprite
5397 } else if bg_idx != 0 {
5398 PixelLayer::Background
5399 } else {
5400 PixelLayer::Backdrop
5401 };
5402 let rec = PixelProvenance {
5403 scanline: self.scanline,
5404 dot: self.dot,
5405 layer,
5406 palette_addr: pal_addr,
5407 palette_index: u8::try_from(palette_index(pal_addr)).unwrap_or(0),
5408 color: final_idx,
5409 color_mask: self.mask.bits() & 0xE1,
5410 // The DISPLAYED tile's addresses, from the cascade — `v` has
5411 // already advanced two tiles past this pixel.
5412 nt_addr: self.prov_bg_cur.nt,
5413 at_addr: self.prov_bg_cur.at,
5414 pattern_addr: match layer {
5415 PixelLayer::Sprite => self.prov_spr_addr[spr_slot],
5416 PixelLayer::Background => self.prov_bg_cur.pattern,
5417 PixelLayer::Backdrop => PATTERN_ADDR_NONE,
5418 },
5419 bg_idx,
5420 bg_pal,
5421 spr_idx,
5422 spr_pal,
5423 sprite_slot: if shows_sprite {
5424 u8::try_from(spr_slot).unwrap_or(0)
5425 } else {
5426 crate::provenance::SPRITE_SLOT_NONE
5427 },
5428 sprite_front: spr_priority_front,
5429 sprite_zero: spr_zero_pixel,
5430 fine_x: fx,
5431 fine_y: u8::try_from((self.v >> 12) & 7).unwrap_or(0),
5432 };
5433 if let Some(frame) = self.prov_frame.as_mut() {
5434 frame.set(pixel_x as usize, pixel_y as usize, rec);
5435 }
5436 }
5437
5438 // Decrement sprite X-counters / shift sprite shift regs.
5439 //
5440 // v2.0 (ppu-sprite-shifter-counter): the X-COUNTER decrements every
5441 // visible dot regardless of rendering (Stale Sprite test 2 — forced
5442 // blank does NOT halt the counters), but the SHIFTER only advances while
5443 // rendering is ENABLED on the 1-PPU-dot-delayed gate (`rendering_enabled_
5444 // delayed`, the same gate `shift_bg` uses — test 3: the shifter PAUSES in
5445 // forced blank so a sprite's data survives a long blank and still draws
5446 // on re-enable). `spr_halted` is the persistent latch (set at counter==0
5447 // or across a disable; re-armed at dot 339) carrying Stale Sprite t5/6.
5448 // Default build: the legacy unconditional shift.
5449 for i in 0..self.spr_count as usize {
5450 if self.spr_halted[i] || self.spr_x[i] == 0 {
5451 // Halted / drawing: latch and shift while rendering is enabled.
5452 // The `spr_x == 0` term is load-bearing — a slot re-armed at dot
5453 // 339 with the counter already 0 must SHIFT this dot (not just
5454 // latch) to match the legacy `spr_x == 0 => shift` timing.
5455 self.spr_halted[i] = true;
5456 if self.rendering_enabled_delayed {
5457 self.spr_shift_lo[i] <<= 1;
5458 self.spr_shift_hi[i] <<= 1;
5459 }
5460 } else {
5461 // Counting: decrement every visible dot (forced blank does not
5462 // halt — test 2). On reaching 0, halt this tick.
5463 self.spr_x[i] -= 1;
5464 if self.spr_x[i] == 0 {
5465 self.spr_halted[i] = true;
5466 }
5467 }
5468 }
5469 // v2.9.5: the re-arm that the odd-frame skip deferred takes effect now, after
5470 // pixel 0 was drawn and shifted in the drawing state. The counters then
5471 // start one dot late, which is what keeps pixels 1-7 in place.
5472 if self.spr_rearm_deferred {
5473 self.spr_rearm_deferred = false;
5474 for i in 0..self.spr_count as usize {
5475 self.spr_halted[i] = false;
5476 }
5477 }
5478 }
5479
5480 // ------------------------------------------------------------------
5481 // Sprite evaluation + tile fetch.
5482 // ------------------------------------------------------------------
5483
5484 /// Per-PPU-dot sprite-evaluation FSM.
5485 ///
5486 /// Reproduces the 2C02's three-phase sprite-eval state machine across
5487 /// dots 1..=256 of every visible scanline and the pre-render line:
5488 ///
5489 /// - **Dot 0**: reset FSM working state.
5490 /// - **Dots 1..=64**: clear secondary OAM to `$FF`. One byte cleared
5491 /// every two dots (32 bytes over 64 dots). Reads of `$2004` during
5492 /// this phase return `$FF` on real hardware.
5493 /// - **Dots 65..=256**: 192 dots = 96 read/write pairs. Odd dots read
5494 /// a byte from primary OAM into a latch; even dots commit the latch
5495 /// into secondary OAM (when copying is enabled). The buggy `n+m`
5496 /// increment for overflow detection (when 8 sprites are already
5497 /// latched) matches the documented hardware quirk that
5498 /// `sprite_overflow_tests/4-Obscure` and `/5-Emulator` exercise.
5499 /// - **Dot 256**: commit `spr_count` and pre-clear unused slot
5500 /// rendering-side arrays so the pixel pipeline never emits stale
5501 /// sprite pixels.
5502 ///
5503 /// The actual per-slot pattern-table fetch (and its A12 transitions)
5504 /// happens later, in [`Self::fetch_sprite_tile`], unchanged. Sprite-
5505 /// tile fetches still dispatch at dots 260, 268, ..., 316 — preserving
5506 /// the canonical "241 A12 rises per NTSC frame" MMC3 IRQ count.
5507 /// v2.0 Tier 1.2 — value `$2004` returns while the screen is being drawn.
5508 ///
5509 /// Mirrors Mesen2 `NesPpu::ReadRam`'s `SpriteData` case
5510 /// (`NesPpu.cpp:361-380`): during the sprite-tile-load window (dots
5511 /// 257-320) the OAM data bus carries `secondary_oam[sprite*4 + min(step,3)]`
5512 /// (the 4th byte held for the 5 idle fetch cycles); at every other rendered
5513 /// dot it carries `oam_bus_copybuffer` (the sprite-eval data latch
5514 /// maintained by [`Self::tick_oam_bus`]). Caller has already checked
5515 /// `scanline <= 239 && rendering`.
5516 /// Is the isolated OAM-data-bus model the thing a `$2004` read observes
5517 /// right now?
5518 ///
5519 /// EXTRACTED so the register read and the diagnostic trace cannot drift.
5520 /// `tick_oam_bus` runs only on visible scanlines with rendering enabled, so
5521 /// off that window `oam_bus_copybuffer` holds whatever the last rendered dot
5522 /// left in it. `cpu_read_register` has always guarded against that; the
5523 /// v2.5.6 state-trace field did not, and would have reported a stale
5524 /// secondary-OAM byte as though it were the `$2004` value for the dot --
5525 /// which defeats the entire reason that field exists.
5526 #[must_use]
5527 pub(crate) const fn oam_data_bus_is_live(&self) -> bool {
5528 self.scanline <= 239 && self.is_render_scanline() && self.mask.rendering_enabled()
5529 }
5530
5531 /// What a `$2004` read would return at this exact dot, model included.
5532 ///
5533 /// This is the diagnostic counterpart of the `$2004` arm in
5534 /// `cpu_read_register`, minus that arm's side effects (open-bus touch, OAM
5535 /// decay refresh) -- a trace must not perturb what it observes.
5536 /// Feature-gated rather than `#[allow(dead_code)]`: its only caller is
5537 /// `build_state_record`, which is itself behind `ppu-state-trace`. An item
5538 /// that is live under one feature and dead by default is the case that
5539 /// earns an attribute — and compiling it out entirely is better than
5540 /// suppressing the warning about it.
5541 #[cfg(feature = "ppu-state-trace")]
5542 #[must_use]
5543 pub(crate) fn oam_data_bus_observed(&self) -> u8 {
5544 if self.oam_data_bus_is_live() {
5545 self.oam_data_bus_read()
5546 } else {
5547 let v = self.oam[self.oam_addr as usize];
5548 if (self.oam_addr & 0x03) == 0x02 {
5549 v & 0xE3
5550 } else {
5551 v
5552 }
5553 }
5554 }
5555
5556 fn oam_data_bus_read(&self) -> u8 {
5557 if (257..=320).contains(&self.dot) {
5558 let phase = (self.dot - 257) % 8;
5559 let step = if phase > 3 { 3 } else { phase };
5560 let oam_addr = ((self.dot - 257) / 8) * 4 + step;
5561 self.oam_bus_secondary[(oam_addr & 0x1F) as usize]
5562 } else {
5563 self.oam_bus_copybuffer
5564 }
5565 }
5566
5567 /// v2.0 Tier 1.2 — per-dot driver for the isolated OAM-data-bus model.
5568 ///
5569 /// A side-effect-free model of the `NESdev`-documented PPU sprite-evaluation
5570 /// sequence (`NESdev` wiki "PPU sprite evaluation" + "PPU rendering"):
5571 /// secondary-OAM clear (dots 1-64), evaluation (65-256), and sprite fetch
5572 /// (257-320) in the default configuration (the optional OAMADDR sprite-eval
5573 /// corruption glitch disabled; the 8-sprite overflow bug is still modeled),
5574 /// plus the cycle-321 copy-buffer reset. It maintains ONLY
5575 /// `oam_bus_copybuffer` +
5576 /// the parallel `oam_bus_secondary`; it reads primary `oam` read-only and
5577 /// NEVER touches the real sprite-eval / overflow / sprite-zero state (so
5578 /// the existing rendering FSM is unperturbed — `$2004` reads are the sole
5579 /// observable effect of this whole feature). Called each dot on visible
5580 /// scanlines (0-239) when rendering is enabled.
5581 /// Maintain `OAM2Address` and the "OAM2 Overflowed" freeze flag.
5582 ///
5583 /// Called from `tick_oam_bus`, so it inherits that call site's gate:
5584 /// visible scanline AND rendering enabled. That gate IS the ROM's
5585 /// "cleared if rendering is enabled during dots 63, 255, and 339"
5586 /// condition, which is why no explicit rendering test appears here --
5587 /// and why a scanline with rendering off correctly PRESERVES both the
5588 /// counter and the flag, which is the whole behaviour under test.
5589 ///
5590 /// Runs before `tick_oam_bus`'s clear- and eval-window early-returns,
5591 /// because two of the three reset dots (63 and 255) fall inside them.
5592 fn tick_oam2_address(&mut self, cycle: u16) {
5593 if matches!(cycle, 63 | 255 | 339) {
5594 self.oam2_fetch_addr = 0;
5595 self.oam2_overflowed = false;
5596 }
5597 if cycle == 257 {
5598 self.oam2_fetch_frozen = self.oam2_overflowed;
5599 }
5600 if (256..=320).contains(&cycle) && (cycle & 1) == 0 && !self.oam2_overflowed {
5601 // 32 bytes across the fetch window = one step per two dots. The
5602 // wrap raises the flag, which suppresses the 33rd increment and
5603 // leaves the address resting at 0 -- see the field docs.
5604 self.oam2_fetch_addr = (self.oam2_fetch_addr + 1) & 0x1F;
5605 if self.oam2_fetch_addr == 0 {
5606 self.oam2_overflowed = true;
5607 }
5608 }
5609 if cycle == 321 {
5610 // After fetch the `$2004` bus rests wherever OAM2Address actually
5611 // STOPPED -- index 0 in the ordinary case, because a complete
5612 // fetch wraps it there. This used to be a hard `[0]`, which is why
5613 // it was right in general and wrong exactly when the counter had
5614 // not completed its wrap.
5615 // `& 0x1F` matches every other `oam_bus_secondary` index site and
5616 // makes the bound structural: the counter is a 5-bit register and
5617 // `tick_oam2_address` already keeps it in range, while a restore
5618 // rejects anything wider. This is the third layer, not the first.
5619 self.oam_bus_copybuffer =
5620 self.oam_bus_secondary[(self.oam2_fetch_addr & 0x1F) as usize];
5621 }
5622 }
5623
5624 fn tick_oam_bus(&mut self) {
5625 let cycle = self.dot;
5626 // v2.3.0 (perf) — take the dot-0 early-out BEFORE deriving the sprite
5627 // height and y-test reference; both were computed unconditionally and
5628 // then discarded on this dot. Byte-identical: neither value is observable
5629 // on the path that returns here.
5630 if cycle == 0 {
5631 return;
5632 }
5633 self.tick_oam2_address(cycle);
5634 // NOTE (v2.3.1 G3): pushing these two below the `cycle < 65` early-out
5635 // as well — they are dead across the dots 1..=64 clear window — was
5636 // measured and produced NO change on any workload across two runs. LLVM
5637 // already sinks pure computations past branches that do not use them.
5638 // Do not re-attempt as a performance change; see `docs/performance.md`.
5639 let sprite_height: i16 = if self.ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
5640 16
5641 } else {
5642 8
5643 };
5644 // Y-test reference: the scanline being evaluated (sprites render on
5645 // scanline+1).
5646 let scan = self.scanline;
5647 if cycle < 65 {
5648 // Secondary-OAM clear (cycles 1-64): the bus carries $FF and the
5649 // parallel secondary OAM is filled with $FF, 1 byte per 2 dots.
5650 self.oam_bus_copybuffer = 0xFF;
5651 self.oam_bus_secondary[((cycle - 1) >> 1) as usize] = 0xFF;
5652 return;
5653 }
5654 if cycle <= 256 {
5655 if cycle & 1 == 1 {
5656 // Odd cycle: read a byte from primary OAM into the bus latch.
5657 if cycle == 65 {
5658 // ProcessSpriteEvaluationStart: seed the eval pointer from
5659 // OAMADDR (eval can begin mid-sprite if $2003 was written).
5660 self.oam_bus_sprite_in_range = false;
5661 self.oam_bus_secondary_addr = 0;
5662 self.oam_bus_overflow_counter = 0;
5663 self.oam_bus_copy_done = false;
5664 self.oam_bus_addr_h = (self.oam_addr >> 2) & 0x3F;
5665 self.oam_bus_addr_l = self.oam_addr & 0x03;
5666 }
5667 let addr = ((self.oam_bus_addr_l & 0x03) | (self.oam_bus_addr_h << 2)) as usize;
5668 // v2.1.4 F2.3 — OAM-decay read hook (no-op at the default): a
5669 // sprite-evaluation primary-OAM read refreshes the row's DRAM
5670 // cells (this is what keeps OAM alive during normal rendering).
5671 self.oam_decay_on_read((addr & 0xFF) as u8);
5672 let raw = self.oam[addr & 0xFF];
5673 // OAM byte 2 (attributes) bits 2-4 are unimplemented (read 0).
5674 self.oam_bus_copybuffer = if addr & 0x03 == 0x02 { raw & 0xE3 } else { raw };
5675 } else {
5676 // Even cycle: copy / decide.
5677 let cb = self.oam_bus_copybuffer as i16;
5678 let cb_in_range = scan >= cb && scan < cb + sprite_height;
5679 if self.oam_bus_copy_done {
5680 self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5681 // OAM write-disable turns secondary-OAM writes into reads.
5682 // On early (pre-rev-G) 2C02s the data bus reads back the
5683 // last byte the OAM-address counter rests on EVEN when fewer
5684 // than 8 sprites were found (secondary_addr < 0x20) — the
5685 // "OAM2[OAM2Address] every other cycle" behavior AccuracyCoin
5686 // `$2004 Stress` section 6 documents. Mesen2 gates this on
5687 // `secondary_addr >= 0x20` (rev-G+), which is why no Mesen
5688 // config reproduces the section-6 `$03`; the test's answer
5689 // key (the spec) wants the unconditional read. Each
5690 // out-of-range sprite's Y was already written to
5691 // `secondary[secondary_addr]` (the frozen index) below, so
5692 // this reads back that last-written Y.
5693 self.oam_bus_copybuffer =
5694 self.oam_bus_secondary[(self.oam_bus_secondary_addr & 0x1F) as usize];
5695 } else {
5696 if !self.oam_bus_sprite_in_range && cb_in_range {
5697 self.oam_bus_sprite_in_range = true;
5698 }
5699 if self.oam_bus_secondary_addr < 0x20 {
5700 // Copy one byte to (parallel) secondary OAM.
5701 self.oam_bus_secondary[self.oam_bus_secondary_addr as usize] =
5702 self.oam_bus_copybuffer;
5703 if self.oam_bus_sprite_in_range {
5704 self.oam_bus_addr_l += 1;
5705 self.oam_bus_secondary_addr += 1;
5706 // OAM2 full: the address overflowed, which raises
5707 // the freeze flag. The dot-255 reset normally
5708 // clears it again before sprite fetch; it survives
5709 // only if rendering is disabled across that dot.
5710 self.oam2_overflowed |= self.oam_bus_secondary_addr == 0x20;
5711 if self.oam_bus_addr_l >= 4 {
5712 self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5713 self.oam_bus_addr_l = 0;
5714 if self.oam_bus_addr_h == 0 {
5715 self.oam_bus_copy_done = true;
5716 }
5717 }
5718 if self.oam_bus_secondary_addr.trailing_zeros() >= 2 {
5719 // Finished copying all 4 bytes of this sprite.
5720 self.oam_bus_sprite_in_range = false;
5721 if self.oam_bus_addr_l != 0 && !cb_in_range {
5722 self.oam_bus_addr_l = 0;
5723 }
5724 }
5725 } else {
5726 // Nothing to copy — skip to the next sprite.
5727 self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5728 self.oam_bus_addr_l = 0;
5729 if self.oam_bus_addr_h == 0 {
5730 self.oam_bus_copy_done = true;
5731 }
5732 }
5733 } else {
5734 // 8 sprites found: secondary-OAM writes become reads.
5735 self.oam_bus_copybuffer =
5736 self.oam_bus_secondary[(self.oam_bus_secondary_addr & 0x1F) as usize];
5737 if self.oam_bus_sprite_in_range {
5738 // Overflow detected. (NOTE: the REAL SpriteOverflow
5739 // flag is owned by the existing eval FSM — this
5740 // isolated model deliberately does not set it.)
5741 self.oam_bus_addr_l += 1;
5742 if self.oam_bus_addr_l == 4 {
5743 self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5744 self.oam_bus_addr_l = 0;
5745 }
5746 if self.oam_bus_overflow_counter == 0 {
5747 self.oam_bus_overflow_counter = 3;
5748 } else {
5749 self.oam_bus_overflow_counter -= 1;
5750 if self.oam_bus_overflow_counter == 0 {
5751 self.oam_bus_copy_done = true;
5752 self.oam_bus_addr_l = 0;
5753 }
5754 }
5755 } else {
5756 // Sprite-eval bug: increment BOTH H and L.
5757 self.oam_bus_addr_h = (self.oam_bus_addr_h + 1) & 0x3F;
5758 self.oam_bus_addr_l = (self.oam_bus_addr_l + 1) & 0x03;
5759 if self.oam_bus_addr_h == 0 {
5760 self.oam_bus_copy_done = true;
5761 }
5762 }
5763 }
5764 }
5765 }
5766 }
5767 }
5768
5769 // v2.3.0 (perf) — called once per ELIGIBLE dot on the fast dot path (visible
5770 // dots 1..=256 with rendering enabled: up to 61,440/frame, not all 89,342 —
5771 // idle lines and rendering-disabled paths bypass it entirely);
5772 // `perf annotate` showed its own prologue/epilogue (`push`/`ret`) as the two
5773 // hottest instructions in the body, i.e. pure call overhead LLVM had declined
5774 // to remove. `inline` lets it be folded into the dot loop. Byte-identical (an
5775 // inlining hint changes no behavior); adopted only if it clears the >3% bar.
5776 #[inline]
5777 pub(crate) fn tick_sprite_eval_per_dot(&mut self) {
5778 // Y-test reference line for sprite evaluation. Per nesdev
5779 // "PPU OAM" (Byte 0): "The first scanline that the sprite is
5780 // rendered on is one greater than this value." Hardware
5781 // performs the y-test `(scanline - y) in [0, h-1]` using the
5782 // CURRENT scanline counter — the eval at scanline N produces
5783 // sprites that render on scanline N+1. So sprite Y=N renders
5784 // on scanlines N+1..=N+h.
5785 //
5786 // Pre-render (scanline 261) prepares for scanline 0, but
5787 // scanline 0 never displays sprites per nesdev. We model
5788 // this by using -1 as the y-test reference, which makes
5789 // `-1 - y < 0` for all OAM y values, so the y-test always
5790 // fails at pre-render and scanline 0 sees no sprites.
5791 //
5792 // That line, and the sprite height, are computed inside the
5793 // `65..=256` arm below, their only use. They are dead on 149 of the
5794 // 341 dots, and computing them up front cost real time.
5795 //
5796 // v2.3.1 measured this sink (G3) as "no change" and this comment
5797 // used to forbid re-trying it. That run used the pre-v2.9.1
5798 // `ab_check.sh`, which timed the same binary on both sides. v2.9.8
5799 // re-measured it with the fixed tool: -1.0% to -3.1% on all four
5800 // frame workloads, shipped `_fast` paths included, in two runs
5801 // (`docs/performance.md`, v2.9.8 campaign).
5802 match self.dot {
5803 0 => {
5804 // Start-of-scanline: reset FSM working state. We do NOT
5805 // touch the rendering-side `spr_*` arrays or
5806 // `spr_zero_in_line` here — they were committed at the
5807 // PREVIOUS scanline's dot 256 and are about to be read
5808 // by this scanline's sprite-pixel evaluator on dots
5809 // 1..=256.
5810 self.sprite_eval_n = 0;
5811 self.sprite_eval_m = 0;
5812 self.sprite_eval_found = 0;
5813 self.sprite_eval_sec_idx = 0;
5814 self.sprite_eval_copying = false;
5815 self.sprite_eval_done = false;
5816 self.sprite_eval_overflow_search = false;
5817 self.sprite_eval_read_latch = 0xFF;
5818 self.sprite_eval_zero_found = false;
5819 // Phase 3a: capture eval base from OAMADDR at the
5820 // dot-0 reset so the dots 65-256 active loop starts
5821 // walking from the captured `(start_n, start_m)`
5822 // position. Mesen2 captures at cycle 65 (in
5823 // ProcessSpriteEvaluationStart); we capture at dot 0
5824 // because our FSM does the eval-base read BEFORE
5825 // dot 65 (the first read at dot 65 already needs
5826 // the offset). This matters when the CPU writes
5827 // $2003 mid-vblank to set OAMADDR before the next
5828 // scanline's eval begins.
5829 {
5830 self.sprite_eval_n = (self.oam_addr >> 2) & 0x3F;
5831 self.sprite_eval_m = self.oam_addr & 0x03;
5832 }
5833 self.sprite_eval_first_iter = true;
5834 }
5835 1..=64 => {
5836 // Clear phase. Even-dot writes a $FF into secondary OAM
5837 // (1 byte per 2 dots, 32 bytes over 64 dots). Odd dots
5838 // are idle reads (driving $FF onto the bus).
5839 //
5840 // The pre-2026-05-17 implementation also reset the
5841 // rendering-side `spr_*` arrays + `spr_count` +
5842 // `spr_zero_in_line` here at dot 64. That was a B8a
5843 // regression: the rendering loop at line 1146..=1220
5844 // READS those arrays on dots 1..=256 of the CURRENT
5845 // scanline, so resetting them mid-scanline destroyed
5846 // sprites for dots 64..=256 (the right ~75% of every
5847 // scanline). The dot 256 End-of-eval fixup below is
5848 // the correct time to commit the NEXT scanline's
5849 // values; the dot 64 reset has been removed.
5850 if (self.dot & 1) == 0 {
5851 let idx = ((self.dot - 1) >> 1) as usize;
5852 if idx < self.secondary_oam.len() {
5853 self.secondary_oam[idx] = 0xFF;
5854 }
5855 }
5856 }
5857 65..=256 => {
5858 if self.dot == 65 {
5859 // v3.1.0: OAMADDR AT TICK 65 sets where evaluation
5860 // starts (nesdev "PPU registers" -> OAMADDR, "Values
5861 // during rendering"), so re-seed `(n, m)` here. The dot-0
5862 // capture above stays as the reset value, but a `$2003`
5863 // write (or a rendering-time `$2004` bump) during the
5864 // dots 1-64 clear must still move the start. Until
5865 // v3.1.0 only the dot-0 value counted, which was
5866 // invisible while every test wrote `$2003` before dot 0:
5867 // AccuracyCoin f5f41dc2 moved its "Misaligned OAM
5868 // behavior" write to dots 28-29 of scanline 0 (one
5869 // `JSR`/`RTS` pair later) and test 3 failed, evaluating
5870 // from the stale address. The OAM-bus model above has
5871 // always seeded at cycle 65.
5872 self.sprite_eval_n = (self.oam_addr >> 2) & 0x3F;
5873 self.sprite_eval_m = self.oam_addr & 0x03;
5874 }
5875 if !self.sprite_eval_done {
5876 let next_line: i16 = if self.scanline == self.region.prerender_line() {
5877 -1
5878 } else {
5879 self.scanline
5880 };
5881 let sprite_height: i16 = if self.ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
5882 16
5883 } else {
5884 8
5885 };
5886 self.tick_sprite_eval_active_dot(next_line, sprite_height);
5887 }
5888
5889 if self.dot == 256 {
5890 // End-of-eval fixup: commit spr_count and the
5891 // eval-side sprite-0 latch onto the rendering-side
5892 // arrays. Pre-clear slots we did NOT fill so unused
5893 // ones produce no output even though
5894 // `fetch_sprite_tile` always runs all 8 slots.
5895 self.spr_count = self.sprite_eval_found;
5896 self.spr_zero_in_line = self.sprite_eval_zero_found;
5897 for i in (self.spr_count as usize)..8 {
5898 self.spr_shift_lo[i] = 0;
5899 self.spr_shift_hi[i] = 0;
5900 self.spr_attr[i] = 0;
5901 self.spr_x[i] = 0xFF;
5902 }
5903 }
5904 }
5905 _ => {
5906 // Dots 257..=340: eval is idle; sprite tile fetches happen
5907 // elsewhere (`fetch_sprite_tile`, scheduled at dots 260,
5908 // 268, ..., 316 from the tick() main path).
5909 }
5910 }
5911 }
5912
5913 /// Per-active-dot helper for the per-PPU-dot FSM. Drives the
5914 /// alternating read/write semantics of dots 65..=256 when eval has
5915 /// not yet exhausted primary OAM or set overflow.
5916 #[allow(clippy::too_many_lines)] // Phase 3a feature-gated branches expand the line count beyond the threshold; refactoring into sub-helpers would require sharing 5+ mutable fields by reference, hurting readability.
5917 fn tick_sprite_eval_active_dot(&mut self, next_line: i16, sprite_height: i16) {
5918 if (self.dot & 1) == 1 {
5919 // Odd dot: read.
5920 // Per nesdev wiki "PPU sprite evaluation": during dots 65-256,
5921 // the hardware updates OAMADDR to track the current eval read
5922 // position. A CPU $2004 read at this time sees the OAM byte
5923 // at that walking index. We surface the eval position into
5924 // `oam_addr` so that CPU reads of $2004 during sprite eval
5925 // observe the same behavior as real silicon. The dot-257-320
5926 // OAMADDR-reset added in `Ppu::tick` washes this back to 0
5927 // after eval, preserving the post-eval semantics that the
5928 // existing $4014 OAM DMA / blargg sprite_hit_tests rely on.
5929 // Phase 3a: under the eval-base-from-OAMADDR feature, the
5930 // y-test address ALWAYS uses `n*4 + m` so a misaligned
5931 // start (`oam_addr & 0x03 != 0` at dot 0) reads the
5932 // appropriate byte of the start sprite as the Y candidate
5933 // (Mesen2 `_spriteAddrL` model). Under the legacy path,
5934 // `m` is reset to 0 between sprites and the y-test always
5935 // reads byte 0; the legacy special-case is preserved for
5936 // bit-exact compatibility.
5937 let addr = ((self.sprite_eval_n as usize) * 4) + (self.sprite_eval_m as usize);
5938 // v2.1.4 F2.3 — OAM-decay read hook (no-op at the default): the legacy
5939 // per-dot sprite-eval read path also refreshes the row it reads, so both
5940 // eval models keep OAM alive identically during rendering.
5941 self.oam_decay_on_read((addr & 0xFF) as u8);
5942 self.sprite_eval_read_latch = self.oam[addr & 0xFF];
5943 // Expose the current eval index via the OAMADDR register
5944 // (truncated to u8 via the `& 0xFF` mask). This is the
5945 // documented hardware behavior — see AccuracyCoin
5946 // `TEST_ArbitrarySpriteZero` sub-test 2's lengthy comment
5947 // explaining the eval / OAMADDR interaction.
5948 self.oam_addr = (addr & 0xFF) as u8;
5949 } else {
5950 // Even dot: write/decide.
5951 let latch = self.sprite_eval_read_latch;
5952 if self.sprite_eval_overflow_search {
5953 // Treat the read byte as a y-coord candidate.
5954 let row = next_line - (latch as i16);
5955 if row >= 0 && row < sprite_height {
5956 self.status.insert(PpuStatus::SPRITE_OVERFLOW);
5957 self.sprite_eval_done = true;
5958 } else {
5959 // Buggy n+m increment: increment BOTH.
5960 self.sprite_eval_m = (self.sprite_eval_m + 1) & 0x03;
5961 if self.sprite_eval_n == 63 {
5962 self.sprite_eval_done = true;
5963 } else {
5964 self.sprite_eval_n += 1;
5965 }
5966 }
5967 } else if self.sprite_eval_copying {
5968 // Copy byte (m == 1, 2, 3) into secondary OAM.
5969 let sec_idx = self.sprite_eval_sec_idx as usize;
5970 if sec_idx < self.secondary_oam.len() {
5971 self.secondary_oam[sec_idx] = latch;
5972 }
5973 self.sprite_eval_sec_idx += 1;
5974 self.sprite_eval_m += 1;
5975 // Phase 3a: under the eval-base feature, continue
5976 // copying until the secondary OAM is aligned to a
5977 // sprite boundary (sec_idx % 4 == 0) — Mesen2's model
5978 // (`_secondaryOamAddr & 0x03 == 0` check at line
5979 // 1062). This handles misaligned start where 4
5980 // sequential reads from `(start_n*4+start_m)` span
5981 // sprite boundaries. Under the legacy path,
5982 // `m == 4` is identical to "sec_idx & 3 == 0" because
5983 // copying always starts at m=1 (after y-test at m=0),
5984 // so they're equivalent in the legacy case.
5985 let copy_done = self.sprite_eval_sec_idx.trailing_zeros() >= 2;
5986 if self.sprite_eval_m == 4 {
5987 self.sprite_eval_m = 0;
5988 self.sprite_eval_n = (self.sprite_eval_n + 1) & 0x3F;
5989 }
5990 if copy_done {
5991 // Finished this sprite. found was already
5992 // incremented when the y-byte landed.
5993 self.sprite_eval_copying = false;
5994 // v3.1.0: the fourth byte copied is the X position, and
5995 // the PPU range-tests it exactly as it tests Y. Out of
5996 // range: OAMADDR += 1 then &= $FC, re-aligning. IN range:
5997 // only += 1, so a misaligned start STAYS misaligned
5998 // (AccuracyCoin "Misaligned OAM behavior" tests 4-7, the
5999 // "+4* behavior ... Only +1 with the X Position" rule,
6000 // stated in the ROM's comments). `m` already holds the
6001 // += 1 (the `m == 4` wrap above covers the aligned
6002 // case, where both rules agree), so only the
6003 // out-of-range case clears it. Until v3.1.0 this cleared
6004 // `m` unconditionally; the ROM's pre-f5f41dc2 fail path
6005 // returned into the test body without popping its return
6006 // address, which recorded that failure as a pass.
6007 let x_row = next_line - (latch as i16);
6008 if !(x_row >= 0 && x_row < sprite_height) {
6009 self.sprite_eval_m = 0;
6010 }
6011 // Under feature: the m==4 wrap above already
6012 // advanced n once. Don't double-increment.
6013 // Under legacy: m never wrapped, so n advances
6014 // here for the first (and only) time.
6015 {
6016 // n was already advanced in the m==4 wrap
6017 // block above; just check terminal conditions.
6018 if self.sprite_eval_found == 8 {
6019 self.sprite_eval_overflow_search = true;
6020 }
6021 if self.sprite_eval_n == 0 {
6022 // n wrapped past 63 to 0 — done.
6023 self.sprite_eval_done = true;
6024 }
6025 }
6026 }
6027 } else {
6028 // Y-test for sprite n.
6029 let row = next_line - (latch as i16);
6030 let in_range = row >= 0 && row < sprite_height;
6031 if in_range && self.sprite_eval_found < 8 {
6032 // Write y into secondary OAM and start copying
6033 // bytes 1..=3 over the next 3 even-dot writes.
6034 let sec_idx = self.sprite_eval_sec_idx as usize;
6035 if sec_idx < self.secondary_oam.len() {
6036 self.secondary_oam[sec_idx] = latch;
6037 }
6038 self.sprite_eval_sec_idx += 1;
6039 // Sprite-zero-hit eligibility: per nesdev wiki +
6040 // Mesen2 (`NesPpu::ProcessSpriteEvaluation` line
6041 // 1040-1044, "If the first Y coordinate we load
6042 // is in range, set the sprite 0 flag — this
6043 // happens even if this isn't actually the first
6044 // sprite in OAM (i.e. because OAMADDR was not 0
6045 // when evaluation started)"), the sprite at the
6046 // eval-start position is sprite-zero IFF its Y
6047 // is in range — NOT "first in-range sprite found".
6048 // If the start sprite is out-of-range, no sprite
6049 // on this scanline is sprite-zero. Under Phase 3a,
6050 // gate on `sprite_eval_first_iter` (the first y-test
6051 // of the scanline); the legacy path keeps the
6052 // canonical `n == 0` check.
6053 let is_first_inrange = self.sprite_eval_first_iter;
6054 if is_first_inrange {
6055 self.sprite_eval_zero_flag_on();
6056 }
6057 self.sprite_eval_found += 1;
6058 self.sprite_eval_copying = true;
6059 // Phase 3a: increment from CURRENT m (handles
6060 // misaligned start where eval began at m != 0).
6061 // Legacy path resets to m=1 (canonical "skip Y,
6062 // copy bytes 1..=3" pattern).
6063 {
6064 self.sprite_eval_m += 1;
6065 if self.sprite_eval_m == 4 {
6066 // The Y byte was the LAST byte of slot `n`
6067 // (evaluation started at m = 3). OAMADDR steps
6068 // on into slot n + 1 and the copy continues:
6069 // the PPU copies four bytes whatever the
6070 // alignment. Until v3.1.0 this ended the copy
6071 // here, putting one byte in secondary OAM
6072 // instead of four (AccuracyCoin "Misaligned OAM
6073 // behavior" test 7, offset 3; masked like test
6074 // 6 by the ROM's pre-f5f41dc2 fail path).
6075 self.sprite_eval_m = 0;
6076 if self.sprite_eval_n == 63 {
6077 self.sprite_eval_done = true;
6078 } else {
6079 self.sprite_eval_n += 1;
6080 }
6081 }
6082 }
6083 } else if in_range && self.sprite_eval_found == 8 {
6084 // Defensive: 9th in-range sprite at the y-tested
6085 // cell. In practice the `found == 8` transition
6086 // happens at the end of copying sprite 7, which
6087 // flips into `overflow_search` mode, so this branch
6088 // is unreachable. Kept for safety.
6089 self.status.insert(PpuStatus::SPRITE_OVERFLOW);
6090 self.sprite_eval_done = true;
6091 } else {
6092 // Not in range: advance to the next sprite, and REALIGN.
6093 //
6094 // AccuracyCoin's README gives the rule for the
6095 // secondary-OAM-NOT-full case: "the OAM address is
6096 // incremented by 4 and bitwise ANDed with $FC" -- so the
6097 // byte index CLEARS rather than being carried. This core
6098 // advanced `n` and left `m` at whatever misaligned value
6099 // evaluation started from.
6100 //
6101 // Invisible whenever OAMADDR is a multiple of four (`m` is
6102 // already 0 at every y-test), so only misaligned OAM can
6103 // observe it -- measured at 114 occurrences in a full
6104 // battery run, with no test's verdict depending on it.
6105 // Pinned directly by
6106 // `misaligned_oam_out_of_range_advance_follows_both_rules`.
6107 //
6108 // The FULL case is the other rule in the same entry --
6109 // "only increment the OAM address by 5" -- and is already
6110 // implemented as the buggy n+m increment in
6111 // `sprite_eval_overflow_search` above.
6112 self.sprite_eval_m = 0;
6113 if self.sprite_eval_n == 63 {
6114 self.sprite_eval_done = true;
6115 } else {
6116 self.sprite_eval_n += 1;
6117 }
6118 }
6119 // Phase 3a: clear the "first-iteration" flag AFTER the
6120 // first y-test fires (regardless of in-range result).
6121 // Per Mesen2 `_cycle == 66` semantics — sprite-zero is
6122 // set only on the FIRST y-test that lands in range,
6123 // and only if it's the FIRST iteration overall.
6124 self.sprite_eval_first_iter = false;
6125 }
6126 }
6127 }
6128
6129 /// Helper: set the per-scanline sprite-zero-in-line flag from the
6130 /// FSM. Sets the EVAL-side latch (`sprite_eval_zero_found`); the
6131 /// rendering-side flag (`spr_zero_in_line`) is committed from this
6132 /// latch at dot 256.
6133 const fn sprite_eval_zero_flag_on(&mut self) {
6134 self.sprite_eval_zero_found = true;
6135 }
6136
6137 /// Arm the OAM-corruption disable edge on a `$2001` write — faithful
6138 /// port of `TriCNES`'s `$2001` write path (`Emulator.cs` lines
6139 /// 9684-9696 / 1740-1755). When rendering was ON before the write and
6140 /// the new mask turns BOTH BG + sprites OFF while on a render line (NOT
6141 /// in vblank), set the disable flags. `_instant` is the
6142 /// data-bus-immediate path (OAM eval observes the disable the same
6143 /// cycle); the non-instant flag is the regular 1-dot-delayed path. The
6144 /// disable edge is captured against the live eval pointer during the
6145 /// dots 1-64 window and committed on re-enable; the `!pending` guard
6146 /// stops a write from re-arming over an already-captured corruption.
6147 const fn arm_oam_corruption_disable(&mut self, was_rendering: bool) {
6148 if was_rendering
6149 && !self.mask.rendering_enabled()
6150 && !self.oam_corruption_pending
6151 && (self.scanline < self.region.vblank_start_line()
6152 || self.scanline == self.region.prerender_line())
6153 {
6154 self.oam_corruption_disabled = true;
6155 self.oam_corruption_disabled_instant = true;
6156 }
6157 }
6158
6159 /// OAM-corruption per-dot driver — faithful port of `TriCNES`'s
6160 /// `PPU_Render_SpriteEvaluation` corruption handling (`Emulator.cs`
6161 /// lines 2664-2762). Called every render-line dot (independent of the
6162 /// rendering gate, so the disable edge is observed even after the
6163 /// sprite-eval FSM stops). Three responsibilities, in `TriCNES` order:
6164 ///
6165 /// 1. **Commit on re-enable.** If rendering is currently enabled and a
6166 /// corruption is pending, apply it (`TriCNES` applies on the first
6167 /// rendered dot once `PPU_Mask_Show*_Instant` is set again). The
6168 /// pre-render-line dot-0 hook in `tick` handles the
6169 /// re-enable-during-vblank case.
6170 /// 2. **Maintain `OAM2Address` across the dots 1-64 clear window.**
6171 /// Reset at dot 1; incremented once per even clear dot, masked to
6172 /// 0x1F — exactly as `TriCNES` drives `OAM2Address` during dots 1-64.
6173 /// 3. **Capture the index at the disable edge.** When the disable
6174 /// flag (`oam_corruption_disabled` / `_instant`, armed by the
6175 /// `$2001` write) is set during the dots 1-64 window on a NON
6176 /// pre-render line, set `oam_corruption_pending` and capture
6177 /// `oam_corruption_index = oam2_addr` (the live secondary-OAM write
6178 /// pointer). The pre-render line is excluded from the capture (it is
6179 /// a read-only eval line for OAM-corruption purposes in `TriCNES`).
6180 fn tick_oam_corruption(&mut self, rendering: bool) {
6181 let pre_render = self.scanline == self.region.prerender_line();
6182
6183 // (1) Commit pending corruption once rendering is (re-)enabled
6184 // during the eval window. The pre-render dot-0 path in `tick`
6185 // covers the re-enable-in-VBlank case separately.
6186 if rendering && self.oam_corruption_pending && !pre_render && self.dot >= 1 {
6187 self.process_oam_corruption();
6188 }
6189
6190 // (2) + (3) only matter inside the dots 1-64 secondary-OAM clear
6191 // window. Outside it the disable flags simply persist until the
6192 // next eval window (or are committed above on re-enable).
6193 if (1..=64).contains(&self.dot) {
6194 if self.dot == 1 {
6195 // TriCNES resets OAM2Address at dot 1 of the eval window.
6196 self.oam2_addr = 0;
6197 }
6198
6199 // Capture the disable edge against the LIVE pointer, on a
6200 // non-pre-render line only (TriCNES: capture is skipped on the
6201 // read-only pre-render eval line).
6202 if (self.oam_corruption_disabled || self.oam_corruption_disabled_instant)
6203 && !pre_render
6204 && !self.oam_corruption_pending
6205 {
6206 self.oam_corruption_pending = true;
6207 self.oam_corruption_index = self.oam2_addr;
6208 }
6209 // The disable arming is single-shot: clear it once observed in
6210 // the eval window (TriCNES clears both flags when it fires).
6211 self.oam_corruption_disabled = false;
6212 self.oam_corruption_disabled_instant = false;
6213
6214 // Advance OAM2Address on even clear dots (mirrors TriCNES's
6215 // `OAM2[OAM2Address] = latch; OAM2Address = (OAM2Address+1) &
6216 // 0x1F` on even cycles of the dots 1-64 clear).
6217 if (self.dot & 1) == 0 {
6218 self.oam2_addr = (self.oam2_addr + 1) & 0x1F;
6219 }
6220 }
6221 }
6222
6223 /// Apply a pending OAM corruption — faithful port of `TriCNES`'s
6224 /// `CorruptOAM` (`Emulator.cs` lines 2635-2651): one OAM "row" of 8
6225 /// bytes is overwritten from row 0, plus the corresponding secondary-OAM
6226 /// byte. The index (`oam_corruption_index`) was captured at the disable
6227 /// edge from the live secondary-OAM eval pointer; index 0x20 wraps to 0.
6228 /// Clears the pending flag.
6229 fn process_oam_corruption(&mut self) {
6230 let mut index = self.oam_corruption_index as usize;
6231 if index == 0x20 {
6232 index = 0;
6233 }
6234 // OAM[index*8 + i] = OAM[i] for i in 0..8 (a no-op when index == 0,
6235 // matching TriCNES — it still runs, copying row 0 onto itself).
6236 let first_eight: [u8; 8] = [
6237 self.oam[0],
6238 self.oam[1],
6239 self.oam[2],
6240 self.oam[3],
6241 self.oam[4],
6242 self.oam[5],
6243 self.oam[6],
6244 self.oam[7],
6245 ];
6246 let dst = index * 8;
6247 if dst + 8 <= self.oam.len() {
6248 self.oam[dst..dst + 8].copy_from_slice(&first_eight);
6249 }
6250 // Also corrupt secondary OAM: OAM2[index] = OAM2[0].
6251 if index < self.secondary_oam.len() {
6252 self.secondary_oam[index] = self.secondary_oam[0];
6253 }
6254 self.oam_corruption_pending = false;
6255 }
6256
6257 /// Fetch one sprite slot's pattern bytes. Always called for all 8
6258 /// slots — for unused slots the secondary-OAM bytes are $FF, producing
6259 /// a dummy fetch that still toggles A12 to the sprite pattern table on
6260 /// real hardware. This is what generates the per-scanline A12 rising
6261 /// edge that MMC3's IRQ counter clocks on.
6262 #[allow(clippy::cast_sign_loss)]
6263 fn fetch_sprite_tile<B: PpuBus>(&mut self, bus: &mut B, slot: usize) {
6264 // Mirrors the y-test convention in `tick_sprite_eval_per_dot`:
6265 // `next_line` is the y-test reference = the CURRENT scanline
6266 // counter (or -1 for pre-render). The fetched row index is
6267 // `next_line - y`, which matches the row that will be
6268 // displayed on `next_line + 1` (the next scanline that
6269 // renders the eval result).
6270 //
6271 // v2.0 (ppu-sprite-shifter-counter): treat the pre-render line as
6272 // scanline `(prerender_line & 0xFF)` (NTSC 261 & 255 = 5) for the
6273 // sprite-tile in-range check, so a sprite whose pixel lands on row 5
6274 // loads into the shifters for scanline 0 (the stale secondary-OAM slots
6275 // filtered by the `load` gate below) — AccuracyCoin "Sprites On Scanline
6276 // 0". Default keeps the `-1` reference (scanline 0 sees no sprites).
6277 let next_line: i16 = if self.scanline == self.region.prerender_line() {
6278 self.region.prerender_line() & 0xFF
6279 } else {
6280 self.scanline
6281 };
6282 let sprite_height: i16 = if self.ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
6283 16
6284 } else {
6285 8
6286 };
6287 // With the "OAM2 Overflowed" flag latched, the OAM2 Address cannot
6288 // increment, so sprite fetch re-reads index 0 for EVERY object --
6289 // eight copies of OAM2[0] instead of eight distinct sprites. This is
6290 // the mechanism AccuracyCoin `Frozen OAM2 Increment` targets.
6291 //
6292 // The freeze path is verified end to end, by probe rather than by
6293 // argument: across a battery run the flag is raised 809 times and
6294 // reaches sprite fetch exactly once (the single construction test 2
6295 // builds), on scanline 196, with `spr_count` = 8, `spr_zero_in_line`
6296 // = true, `secondary_oam[0]` = $C1, and all eight slots loading
6297 // Y/tile/attr/X = $C1/$C1/$C1/$C1.
6298 //
6299 // Test 2 nevertheless still fails, and NOT for any sprite reason. Its
6300 // detector is a sprite-zero hit, which needs an opaque BACKGROUND
6301 // pixel under the sprite, and on scanline 197 the background is opaque
6302 // nowhere in x=190..205 (sprite pixels there: 51; background: 0).
6303 // `v` is one vertical increment ahead: fine-Y reads 3 where the test
6304 // needs 2, so the tile it placed for the hit sits a row off. The cause
6305 // is upstream of everything here -- a `$2001` rendering-ENABLE landing
6306 // on dot 256 takes effect one dot early, firing the dot-256 vertical
6307 // increment that hardware does not. The ROM says so in as many words:
6308 // "Rendering is enabled on dot 256, but the PPU's vertical scroll is
6309 // NOT incremented."
6310 //
6311 // That is a `$2001` write-timing gap, not a sprite-evaluation one, and
6312 // it is deliberately NOT patched at the dot-256 site: a compensating
6313 // edit there is the shape v2.5.7 recorded, where a wrong phase had
6314 // every window compensating for it. See docs/STATUS.md.
6315 // Unreachable in ordinary rendering: the flag is cleared at dots 63,
6316 // 255 and 339 whenever rendering is enabled, so reaching fetch with
6317 // it set requires rendering to be off across dot 255.
6318 // A frozen OAM2Address means every read during sprite fetch returns
6319 // index 0 -- the SAME byte four times per sprite, not the first
6320 // sprite's four bytes. AccuracyCoin states the end state explicitly:
6321 // an OAM2 of `C1 24 00 FF C0 C0 ...` is "processed as if it was
6322 // C1 C1 C1 C1 ..." for all 32 bytes. Reading `[0..3]` here would give
6323 // Y/tile/attr/X = C1/24/00/FF, which is a different sprite entirely.
6324 let (y_byte, tile, attr, xpos) = if self.oam2_fetch_frozen {
6325 let frozen = self.secondary_oam[0];
6326 (frozen, frozen, frozen, frozen)
6327 } else {
6328 let base = slot * 4;
6329 (
6330 self.secondary_oam[base],
6331 self.secondary_oam[base + 1],
6332 self.secondary_oam[base + 2],
6333 self.secondary_oam[base + 3],
6334 )
6335 };
6336 let y = y_byte as i16;
6337 let in_use = slot < self.spr_count as usize;
6338 let flip_v = (attr & 0x80) != 0;
6339 let flip_h = (attr & 0x40) != 0;
6340
6341 // For unused slots, the row delta isn't meaningful (Y=$FF makes it
6342 // negative or huge) — pin to 0 so the address arithmetic is well
6343 // defined. The only thing that matters here is that the pattern
6344 // address lands in the sprite pattern table, which it does because
6345 // the sprite-table-select bit is set as PPUCTRL bit 3 (8x8 mode)
6346 // or tile bit 0 (8x16 mode); for the cleared $FF tile in 8x16 mode
6347 // bit 0 = 1 picks the $1000 table.
6348 let mut row: u16 = if in_use {
6349 (next_line.wrapping_sub(y)).clamp(0, sprite_height - 1) as u16
6350 } else {
6351 0
6352 };
6353
6354 let (table, tile_idx, in_tile_row) = if sprite_height == 16 {
6355 let table = u16::from(tile & 0x01) << 12;
6356 let mut tindex = tile & 0xFE;
6357 if flip_v && in_use {
6358 row = 15 - row;
6359 }
6360 if row >= 8 {
6361 tindex = tindex.wrapping_add(1);
6362 row -= 8;
6363 }
6364 (table, tindex, row)
6365 } else {
6366 let table = u16::from(self.ctrl.contains(PpuCtrl::SPRITE_PATTERN_HIGH)) << 12;
6367 let r = if flip_v && in_use { 7 - row } else { row };
6368 (table, tile, r)
6369 };
6370
6371 let addr_lo = table | (u16::from(tile_idx) << 4) | in_tile_row;
6372 let addr_hi = addr_lo | 0x08;
6373 self.observe_a12_addr(bus, addr_lo);
6374 // Sprite CHR fetch: route through `ppu_read_sprite` so MMC5
6375 // (and any other mapper with split sprite vs. BG CHR banking)
6376 // can use its sprite-specific bank registers.
6377 let mut lo = bus.ppu_read_sprite(addr_lo);
6378 self.observe_a12_addr(bus, addr_hi);
6379 let mut hi = bus.ppu_read_sprite(addr_hi);
6380 // W2 ($2007 Stress): stash the RAW (pre-h-flip) pattern bytes so the
6381 // per-dot sprite-fetch read cadence (`tick_sprite_fetch_read`) can
6382 // feed `render_data_bus` for the deferred `$2007` PPUDATA reload.
6383 {
6384 self.spr_fetch_lo_raw[slot] = lo;
6385 self.spr_fetch_hi_raw[slot] = hi;
6386 }
6387 // v2.0 (ppu-sprite-shifter-counter): gate the shifter load on the sprite
6388 // being in-range of `next_line`. On visible scanlines this is a no-op
6389 // (the eval already guarantees every `in_use` slot is in-range), but on
6390 // the pre-render line (feature on, `next_line = 5`) it filters the STALE
6391 // secondary-OAM slots so only sprites whose pixel lands on row 5 load for
6392 // scanline 0. Default (feature off): load every `in_use` slot.
6393 let load = in_use && {
6394 let r = next_line.wrapping_sub(y);
6395 r >= 0 && r < sprite_height
6396 };
6397 if load {
6398 if flip_h {
6399 lo = reverse_bits(lo);
6400 hi = reverse_bits(hi);
6401 }
6402 self.spr_shift_lo[slot] = lo;
6403 self.spr_shift_hi[slot] = hi;
6404 self.spr_attr[slot] = attr;
6405 self.spr_x[slot] = xpos;
6406 // v1.2.0 C3 (hd-pack): stash the 16-byte tile base (in-tile row
6407 // masked off) for this sprite slot so `emit_pixel` can name the
6408 // CHR tile. Telemetry only; no new VRAM read here.
6409 #[cfg(feature = "hd-pack")]
6410 {
6411 self.hd_spr_addr[slot] = addr_lo & 0x1FF0;
6412 self.hd_spr_x[slot] = xpos;
6413 // `in_tile_row` is the post-flip-V fetch row, i.e. exactly the
6414 // (unflipped) replacement texel row to sample.
6415 self.hd_spr_off_y[slot] = u8::try_from(in_tile_row & 0x07).unwrap_or(0);
6416 // CHR-ROM absolute tile index for the sprite, or the CHR-RAM
6417 // sentinel (the common mappers share BG/sprite CHR banking).
6418 self.hd_spr_idx[slot] = bus.chr_phys(addr_lo).map_or(HD_CHR_RAM, |o| o / 16);
6419 }
6420 // v2.3.2 "Lucid": the sprite's pattern ROW address (in-tile row
6421 // KEPT, unlike the `hd-pack` tile base above). Captured separately
6422 // so neither feature's telemetry depends on the other being on.
6423 #[cfg(feature = "debug-hooks")]
6424 {
6425 self.prov_spr_addr[slot] = addr_lo;
6426 }
6427 } else {
6428 #[cfg(feature = "hd-pack")]
6429 {
6430 self.hd_spr_addr[slot] = HD_TILE_NONE;
6431 }
6432 #[cfg(feature = "debug-hooks")]
6433 {
6434 self.prov_spr_addr[slot] = crate::provenance::PATTERN_ADDR_NONE;
6435 }
6436 }
6437 // Else: shift regs already cleared in tick_sprite_eval_per_dot.
6438
6439 // v3.1.0 (`T-SPRITE-LIMIT`): after the eighth REAL fetch, the extra
6440 // sprites for the same line. A no-op unless the option is on.
6441 if slot == 7 {
6442 self.fetch_extra_sprites(bus, next_line, sprite_height);
6443 }
6444 }
6445
6446 /// v3.1.0 (`T-SPRITE-LIMIT`, FE-02): collect and fetch the sprites beyond
6447 /// the eighth for the next scanline, for display only.
6448 ///
6449 /// Runs once per line, after the eighth real sprite fetch, and only when
6450 /// the option is on, evaluation found eight (so the hardware dropped
6451 /// some), the line is visible, and the board's CHR reads are pure
6452 /// ([`PpuBus::chr_reads_are_pure`]: MMC2 / MMC4 latch on CHR reads, the
6453 /// J.Y. ASIC clocks an IRQ on them, and two boards latch address bits, so
6454 /// on those the option draws eight as stock). The fetch calls neither
6455 /// `observe_a12_addr` nor anything else a mapper can see beyond the read,
6456 /// so A12, mapper IRQs and every emulated byte stay exactly stock.
6457 ///
6458 /// Which sprites: an aligned walk of primary OAM from entry 0, skipping
6459 /// the first eight in range (the ones the hardware draws when evaluation
6460 /// starts at OAMADDR 0, as it does on every normally rendered line). A line
6461 /// whose evaluation starts misaligned (a mid-frame `$2003` write, a test
6462 /// construction) can draw a slightly different set; it is a display
6463 /// enhancement, not hardware behaviour. The walk reads `oam` directly and
6464 /// deliberately bypasses the optional OAM-decay read hook, so drawing the
6465 /// extra sprites can never refresh a decaying DRAM row the game could
6466 /// later observe.
6467 fn fetch_extra_sprites<B: PpuBus>(&mut self, bus: &mut B, next_line: i16, height: i16) {
6468 self.spr_extra_count = 0;
6469 if !self.sprite_limit_disabled
6470 || self.spr_count < 8
6471 || !(0..240).contains(&self.scanline)
6472 || !bus.chr_reads_are_pure()
6473 {
6474 return;
6475 }
6476 let mut in_range = 0usize;
6477 for n in 0..64usize {
6478 let y = i16::from(self.oam[n * 4]);
6479 let row = next_line.wrapping_sub(y);
6480 if row < 0 || row >= height {
6481 continue;
6482 }
6483 in_range += 1;
6484 if in_range <= 8 {
6485 continue;
6486 }
6487 let count = usize::from(self.spr_extra_count);
6488 if count == MAX_EXTRA_SPRITES {
6489 break;
6490 }
6491 let tile = self.oam[n * 4 + 1];
6492 let attr = self.oam[n * 4 + 2] & 0xE3;
6493 let x = self.oam[n * 4 + 3];
6494 let flip_v = attr & 0x80 != 0;
6495 #[allow(clippy::cast_sign_loss)] // `row` is in 0..height, checked above
6496 let mut r = row as u16;
6497 let (table, tile_idx) = if height == 16 {
6498 if flip_v {
6499 r = 15 - r;
6500 }
6501 let base = tile & 0xFE;
6502 let idx = if r >= 8 { base.wrapping_add(1) } else { base };
6503 r &= 7;
6504 (u16::from(tile & 0x01) << 12, idx)
6505 } else {
6506 if flip_v {
6507 r = 7 - r;
6508 }
6509 (
6510 u16::from(self.ctrl.contains(PpuCtrl::SPRITE_PATTERN_HIGH)) << 12,
6511 tile,
6512 )
6513 };
6514 let addr = table | (u16::from(tile_idx) << 4) | r;
6515 let mut lo = bus.ppu_read_sprite(addr);
6516 let mut hi = bus.ppu_read_sprite(addr | 0x08);
6517 if attr & 0x40 != 0 {
6518 lo = reverse_bits(lo);
6519 hi = reverse_bits(hi);
6520 }
6521 self.spr_extra_lo[count] = lo;
6522 self.spr_extra_hi[count] = hi;
6523 self.spr_extra_attr[count] = attr;
6524 self.spr_extra_x[count] = x;
6525 self.spr_extra_count += 1;
6526 }
6527 }
6528
6529 fn advance_dot(&mut self) {
6530 // Count every PPU master cycle (one per dot processed) for the NES_NTSC
6531 // colour phase. Output-only / cosmetic; never gates emulation.
6532 self.dot_counter = self.dot_counter.wrapping_add(1);
6533
6534 // Odd-frame skip: when the frame is odd and rendering is enabled,
6535 // the pre-render scanline 261 dot 339 transitions to (0, 0)
6536 // immediately, skipping dot 340.
6537 //
6538 // The rendering check reads `mask_for_skip_check` (two-stage
6539 // pipeline of `mask`, shifted at the bottom of this function), not
6540 // `mask` directly. The two-PPU-clock visibility delay between a
6541 // `$2001` write and this check is what makes blargg
6542 // `ppu_vbl_nmi/10-even_odd_timing` pass: lockstep applies the
6543 // PPUMASK write at the *start* of a CPU cycle, while real hardware
6544 // latches at φ2 (end of cycle). Without the delay the dot-339 skip
6545 // detector observes the write up to two PPU clocks earlier than
6546 // hardware does, mispredicting the skip when the write straddles
6547 // dot 339.
6548 if self.scanline == self.region.prerender_line()
6549 && self.dot == 339
6550 && (self.frame & 1) == 1
6551 && self.skip_gate_mask().rendering_enabled()
6552 && self.region == PpuRegion::Ntsc
6553 {
6554 self.dot = 0;
6555 self.scanline = 0;
6556 self.frame = self.frame.wrapping_add(1);
6557 self.frame_complete = true;
6558 self.snapshot_ntsc_phase();
6559 self.dot0_replaced = true;
6560 // The skipped dot is where the loaded shifters would have seen the
6561 // dot-339 re-arm, so it has not happened yet: they enter scanline 0
6562 // drawing, and `emit_pixel` releases them after pixel 0 (see
6563 // `spr_rearm_deferred`). The skip requires rendering on, so the
6564 // 339 re-arm ran for these slots and this puts them back.
6565 if self.spr_count > 0 {
6566 for i in 0..self.spr_count as usize {
6567 self.spr_halted[i] = true;
6568 }
6569 self.spr_rearm_deferred = true;
6570 }
6571 self.mask_for_skip_check = self.mask_skip_pipe1;
6572 self.mask_skip_pipe1 = self.mask;
6573 return;
6574 }
6575
6576 self.dot += 1;
6577 if self.dot > 340 {
6578 self.dot = 0;
6579 // Advance scanline.
6580 if self.scanline == self.region.prerender_line() {
6581 self.scanline = 0;
6582 self.frame = self.frame.wrapping_add(1);
6583 self.frame_complete = true;
6584 self.snapshot_ntsc_phase();
6585 } else if self.extra_scanlines != 0 && self.scanline + 1 == self.region.prerender_line()
6586 {
6587 // v1.7.0 F3 — PPU extra-scanlines overclock. The line just
6588 // before pre-render is a pure idle vblank line (not visible,
6589 // not the VBL-set line, not pre-render): repeating it emits no
6590 // pixels, sets/clears no flags, and fires no VBL/NMI/A12 event
6591 // — it only adds CPU run-time (the surrounding scheduler still
6592 // clocks the CPU every third dot). When the counter is exhausted
6593 // we fall through to the pre-render line as usual. This whole
6594 // branch is unreachable while `extra_scanlines == 0`, so the
6595 // default build is byte-identical.
6596 if self.extra_lines_remaining == 0 {
6597 self.extra_lines_remaining = self.extra_scanlines;
6598 }
6599 self.extra_lines_remaining -= 1;
6600 if self.extra_lines_remaining == 0 {
6601 // Done inserting: advance to pre-render as normal.
6602 self.scanline += 1;
6603 }
6604 // else: hold on this idle line and run it again.
6605 } else {
6606 self.scanline += 1;
6607 }
6608 }
6609 self.mask_for_skip_check = self.mask_skip_pipe1;
6610 self.mask_skip_pipe1 = self.mask;
6611 }
6612
6613 /// Whether the dot-256 vertical increment rides the shared `rendering_gate`,
6614 /// which is the shipped behaviour.
6615 ///
6616 /// `u8::MAX` means "follow the shared gate"; any other value gives the
6617 /// increment its own depth in the shared history.
6618 #[cfg(feature = "phi2-write-sweep")]
6619 fn scroll_gate_follows_render_gate() -> bool {
6620 SCROLL_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) == u8::MAX
6621 }
6622
6623 /// Shipped build: always the shared gate, folded away.
6624 #[cfg(not(feature = "phi2-write-sweep"))]
6625 const fn scroll_gate_follows_render_gate() -> bool {
6626 true
6627 }
6628
6629 /// Whether the dot-339 sprite-counter re-arm rides the shared
6630 /// `rendering_gate`, which is the shipped behaviour.
6631 ///
6632 /// `SPRITE_REARM_LAG` uses `u8::MAX` for "follow the shared gate" rather
6633 /// than a depth, because the shipped wiring is not any depth of the history
6634 /// -- it is whatever `RENDER_GATE_LAG` resolved to this dot.
6635 #[cfg(feature = "phi2-write-sweep")]
6636 fn sprite_rearm_follows_render_gate() -> bool {
6637 SPRITE_REARM_LAG.load(core::sync::atomic::Ordering::Relaxed) == u8::MAX
6638 }
6639
6640 /// Shipped build: always the shared gate, folded away.
6641 #[cfg(not(feature = "phi2-write-sweep"))]
6642 const fn sprite_rearm_follows_render_gate() -> bool {
6643 true
6644 }
6645
6646 /// The mask the OAM2 counter's gate consults, at the swept depth. Depth 0
6647 /// is the live mask -- the shipped read -- so the default build is
6648 /// unchanged by construction.
6649 #[cfg(feature = "phi2-write-sweep")]
6650 fn oam2_gate_rendering(&self) -> bool {
6651 let depth = OAM2_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) as usize;
6652 if depth == 0 {
6653 self.mask.rendering_enabled()
6654 } else {
6655 self.sweep_mask_history[(depth - 1).min(self.sweep_mask_history.len() - 1)]
6656 .rendering_enabled()
6657 }
6658 }
6659
6660 /// Shipped build: the OAM2 counter carries NO gate of its own.
6661 ///
6662 /// v2.6.18. The call site sits inside the shared `render_line &&
6663 /// rendering_gate` block, so the one-dot-delayed gate is already applied
6664 /// and an additional conjunct could only make the counter's edges differ
6665 /// from every other consumer's. It used to conjoin the LIVE mask, which
6666 /// did exactly that on the DISABLE edge -- see the call site. Depth 2 of
6667 /// the swept history is this same value, which is why
6668 /// [`OAM2_GATE_LAG`]'s default is 2 rather than 0.
6669 // `&self` is unused BY DESIGN and cannot be dropped: the signature has to
6670 // match the `phi2-write-sweep` variant above, which reads the swept mask
6671 // history, so the call site stays identical across both builds. Making this
6672 // an associated function would fork the call site, which is how the two
6673 // builds start disagreeing about something other than the knob.
6674 #[allow(clippy::unused_self)]
6675 #[cfg(not(feature = "phi2-write-sweep"))]
6676 const fn oam2_gate_rendering(&self) -> bool {
6677 true
6678 }
6679
6680 /// The mask the odd-frame-skip gate consults, at the swept pipeline depth.
6681 /// Depth 2 returns `mask_for_skip_check` -- the shipped value -- so the
6682 /// default build is unchanged by construction rather than by claim.
6683 #[cfg(feature = "phi2-write-sweep")]
6684 fn skip_gate_mask(&self) -> PpuMask {
6685 match SKIP_GATE_LAG.load(core::sync::atomic::Ordering::Relaxed) {
6686 0 => self.mask,
6687 1 => self.mask_skip_pipe1,
6688 _ => self.mask_for_skip_check,
6689 }
6690 }
6691
6692 /// Shipped build: the two-stage value, inlined to the same load.
6693 #[cfg(not(feature = "phi2-write-sweep"))]
6694 const fn skip_gate_mask(&self) -> PpuMask {
6695 self.mask_for_skip_check
6696 }
6697
6698 const fn is_render_scanline(&self) -> bool {
6699 // Visible (0..=239) and pre-render line.
6700 self.scanline >= 0 && self.scanline <= self.region.last_visible_line()
6701 || self.scanline == self.region.prerender_line()
6702 }
6703}
6704
6705/// Resolve an address in `$3F00-$3FFF` to a palette RAM index, applying the
6706/// `$3F10/$14/$18/$1C → $3F00/$04/$08/$0C` mirror.
6707const fn palette_index(addr: u16) -> usize {
6708 let mut idx = (addr & 0x1F) as usize;
6709 if matches!(idx, 0x10 | 0x14 | 0x18 | 0x1C) {
6710 idx -= 0x10;
6711 }
6712 idx
6713}
6714
6715/// Reverse the bit order of a byte (used for horizontally-flipped sprites).
6716const fn reverse_bits(b: u8) -> u8 {
6717 b.reverse_bits()
6718}
6719
6720#[cfg(test)]
6721mod tests {
6722 use super::*;
6723
6724 // T-73-005 / T-73-006 (Phase 7): pin the per-region timing table so an
6725 // accidental edit to a region constant trips a test instead of silently
6726 // mis-timing PAL/Dendy. The runtime frame-structure consequences are
6727 // gated by the integration test in
6728 // `crates/rustynes-test-harness/tests/region_timing.rs`.
6729 #[test]
6730 fn ppu_region_constants_match_hardware() {
6731 // NTSC: 262 lines (pre-render 261), VBL@241, no odd-frame skip caveat.
6732 assert_eq!(PpuRegion::Ntsc.prerender_line(), 261);
6733 assert_eq!(PpuRegion::Ntsc.vblank_start_line(), 241);
6734 assert_eq!(PpuRegion::Ntsc.post_reset_mask_cycles(), 29_658);
6735 // PAL: 312 lines (pre-render 311), VBL@241, longer reset mask.
6736 assert_eq!(PpuRegion::Pal.prerender_line(), 311);
6737 assert_eq!(PpuRegion::Pal.vblank_start_line(), 241);
6738 assert_eq!(PpuRegion::Pal.post_reset_mask_cycles(), 33_132);
6739 // Dendy: 312 lines, but VBL starts at 291 (the distinguishing trait).
6740 assert_eq!(PpuRegion::Dendy.prerender_line(), 311);
6741 assert_eq!(PpuRegion::Dendy.vblank_start_line(), 291);
6742 assert_eq!(PpuRegion::Dendy.post_reset_mask_cycles(), 33_132);
6743 // Last visible line is 239 in every region.
6744 for r in [PpuRegion::Ntsc, PpuRegion::Pal, PpuRegion::Dendy] {
6745 assert_eq!(r.last_visible_line(), 239);
6746 }
6747 }
6748
6749 /// `AccuracyCoin`'s README states two rules for advancing `OAMADDR` when a
6750 /// sprite's Y is out of range during evaluation, and they differ by
6751 /// whether secondary OAM is already full:
6752 ///
6753 /// > "the OAM address is incremented by 4 and bitwise ANDed with `$FC`"
6754 ///
6755 /// > "If Secondary OAM is full ... you should instead only increment the
6756 /// > OAM address by 5."
6757 ///
6758 /// Both are invisible while `OAMADDR` is a multiple of four, because the
6759 /// byte index is already 0 at every y-test. They are only observable under
6760 /// MISALIGNED OAM, and measurement showed the whole corpus reaches that
6761 /// case just 114 times in a full `AccuracyCoin` run while no test's verdict
6762 /// depends on it — so this pins it directly instead.
6763 #[test]
6764 fn misaligned_oam_out_of_range_advance_follows_both_rules() {
6765 /// Drive one y-test from a misaligned `OAMADDR` against a Y that is
6766 /// out of range, and report the resulting `(n, m)`.
6767 fn out_of_range_advance(full: bool) -> (u8, u8) {
6768 let mut ppu = Ppu::new(PpuRegion::Ntsc);
6769 ppu.mask = PpuMask::SHOW_SPRITE;
6770 ppu.scanline = 10;
6771 // Misaligned start: OAMADDR $05 seeds n = 1, m = 1.
6772 ppu.oam_addr = 0x05;
6773 // The byte the y-test reads is OAM[n*4 + m] = OAM[5]. Make it far
6774 // out of range for scanline 10.
6775 ppu.oam[5] = 0xF0;
6776
6777 // Dot 0 resets the FSM and captures the eval base from OAMADDR.
6778 ppu.dot = 0;
6779 ppu.tick_sprite_eval_per_dot();
6780 assert_eq!(
6781 (ppu.sprite_eval_n, ppu.sprite_eval_m),
6782 (1, 1),
6783 "seeded misaligned"
6784 );
6785
6786 if full {
6787 // Secondary OAM full: evaluation is in overflow-search mode,
6788 // which is the state the "+5" rule describes.
6789 ppu.sprite_eval_found = 8;
6790 ppu.sprite_eval_overflow_search = true;
6791 }
6792
6793 // Odd dot reads the byte; the following even dot runs the y-test.
6794 ppu.dot = 65;
6795 ppu.tick_sprite_eval_per_dot();
6796 ppu.dot = 66;
6797 ppu.tick_sprite_eval_per_dot();
6798 (ppu.sprite_eval_n, ppu.sprite_eval_m)
6799 }
6800
6801 // NOT full: +4 then AND $FC -- n advances and the byte index CLEARS.
6802 assert_eq!(
6803 out_of_range_advance(false),
6804 (2, 0),
6805 "secondary OAM not full: OAMADDR += 4 & $FC, so the misaligned byte \
6806 index must be cleared, not carried"
6807 );
6808
6809 // FULL: +5 -- n AND m both advance (the classic overflow bug).
6810 assert_eq!(
6811 out_of_range_advance(true),
6812 (2, 2),
6813 "secondary OAM full: OAMADDR += 5, so both the sprite index and the \
6814 byte index advance"
6815 );
6816 }
6817
6818 /// v3.1.0 — the three misaligned-evaluation rules the `AccuracyCoin`
6819 /// `f5f41dc2` re-sync exposed, each pinned on its own so a regression
6820 /// names the rule it broke:
6821 ///
6822 /// 1. evaluation starts at OAMADDR **as of dot 65**, not dot 0 (nesdev
6823 /// "PPU registers" -> OAMADDR), so a write during the clear counts;
6824 /// 2. an in-range Y copies **four** bytes whatever the alignment, also from
6825 /// `m = 3`, where the copy crosses into slot `n + 1`;
6826 /// 3. the fourth byte (X) is range-tested: in range, OAMADDR only steps by
6827 /// one and stays misaligned; out of range, it steps and ANDs with `$FC`.
6828 ///
6829 /// Each assertion failed against the pre-v3.1.0 FSM (dot-0 seed, a
6830 /// one-byte copy from `m = 3`, an unconditional realign).
6831 #[test]
6832 fn misaligned_oam_eval_starts_at_dot_65_copies_four_bytes_and_tests_x() {
6833 /// A PPU on scanline 10 whose dot-0 reset has already run with
6834 /// OAMADDR 0, so only a later seed can pick up `oam_addr`.
6835 fn ppu_after_dot0(oam_addr: u8, oam: &[(usize, u8)]) -> Ppu {
6836 let mut ppu = Ppu::new(PpuRegion::Ntsc);
6837 ppu.mask = PpuMask::SHOW_SPRITE;
6838 ppu.scanline = 10;
6839 ppu.oam.fill(0xFF);
6840 for &(i, v) in oam {
6841 ppu.oam[i] = v;
6842 }
6843 ppu.oam_addr = 0;
6844 ppu.dot = 0;
6845 ppu.tick_sprite_eval_per_dot();
6846 // The write lands during the clear, after the dot-0 reset.
6847 ppu.oam_addr = oam_addr;
6848 ppu
6849 }
6850 fn run_dots(ppu: &mut Ppu, from: u16, to: u16) {
6851 for d in from..=to {
6852 ppu.dot = d;
6853 ppu.tick_sprite_eval_per_dot();
6854 }
6855 }
6856
6857 // (1) Seed at dot 65: OAMADDR 2 written after dot 0. Y at OAM[2] is in
6858 // range for scanline 10 (Y = 8), and the walk must start there.
6859 let mut ppu = ppu_after_dot0(0x02, &[(2, 8), (3, 0x11), (4, 0x22), (5, 0x33)]);
6860 run_dots(&mut ppu, 65, 66);
6861 assert_eq!(
6862 ppu.secondary_oam[0], 8,
6863 "evaluation must read its first Y from OAMADDR as of dot 65 (OAM[2])"
6864 );
6865
6866 // (2) Four bytes from m = 3: OAM[3] is Y, OAM[4..=6] belong to slot 1.
6867 let mut ppu = ppu_after_dot0(0x03, &[(3, 8), (4, 0xA1), (5, 0xA2), (6, 0xA3)]);
6868 run_dots(&mut ppu, 65, 72);
6869 assert_eq!(
6870 ppu.secondary_oam[..4],
6871 [8, 0xA1, 0xA2, 0xA3],
6872 "a misaligned in-range sprite copies four bytes, across the slot edge"
6873 );
6874
6875 // (3) X range test, from OAMADDR 1: Y = OAM[1], X = OAM[4].
6876 let x_case = |x: u8| {
6877 let mut ppu = ppu_after_dot0(0x01, &[(1, 8), (2, 0x11), (3, 0x22), (4, x)]);
6878 run_dots(&mut ppu, 65, 72);
6879 u16::from(ppu.sprite_eval_n) * 4 + u16::from(ppu.sprite_eval_m)
6880 };
6881 assert_eq!(
6882 x_case(8),
6883 0x05,
6884 "X in range: OAMADDR += 1 only, so the walk stays misaligned at $05"
6885 );
6886 assert_eq!(
6887 x_case(0xF0),
6888 0x04,
6889 "X out of range: OAMADDR += 1 then & $FC, realigning to $04"
6890 );
6891 }
6892
6893 /// v2.6.18 — the depth-2 rendering-gate pipeline must SHIFT, not freeze.
6894 ///
6895 /// Drives the named pair directly rather than `tick`, so it needs no bus
6896 /// and — the point — never stores `RENDER_GATE_LAG`, which 13 other tests
6897 /// in this binary would observe concurrently.
6898 ///
6899 /// Against the pre-fix code (`render_gate_prev2 = self.
6900 /// rendering_enabled_delayed` at tick-end, read back AFTER the depth-2
6901 /// re-point had already overwritten it) the field is assigned to itself and
6902 /// the gate reads `false` for every dot of the run. That made every
6903 /// `lag >= 2` cell of the derivation sweep a measurement of a permanently
6904 /// disabled gate. The assertions below fail on the first `true`.
6905 #[cfg(feature = "phi2-write-sweep")]
6906 #[test]
6907 fn render_gate_lag_shifts_a_two_dot_pipeline() {
6908 let mut ppu = Ppu::new(PpuRegion::Ntsc);
6909 // Power-on: both stages clear, rendering off.
6910 assert!(!ppu.rendering_enabled_delayed);
6911 assert!(!ppu.render_gate_prev2);
6912
6913 // One dot of the pipeline at depth 2, returning the value this dot's
6914 // gate reads. Mirrors `tick`'s order exactly: re-point, then fill
6915 // stage N-1 from the live mask, then shift N-1 into N-2.
6916 let dot = |ppu: &mut Ppu, rendering: bool| -> bool {
6917 ppu.mask = if rendering {
6918 PpuMask::SHOW_BG
6919 } else {
6920 PpuMask::empty()
6921 };
6922 let prev1 = ppu.render_gate_begin_dot(2);
6923 let gate = ppu.rendering_enabled_delayed;
6924 // `tick` shifts BEFORE it refills stage N-1; keep that order, or
6925 // the mutation below stops reproducing the shipped defect.
6926 ppu.render_gate_end_dot(prev1);
6927 ppu.rendering_enabled_delayed = ppu.mask.rendering_enabled();
6928 gate
6929 };
6930
6931 // Rendering goes ON and stays on: the gate must follow two dots later.
6932 assert!(!dot(&mut ppu, true), "dot 1: gate still sees 2 dots ago");
6933 assert!(!dot(&mut ppu, true), "dot 2: gate still sees 2 dots ago");
6934 assert!(
6935 dot(&mut ppu, true),
6936 "dot 3: the ON edge has shifted through"
6937 );
6938 assert!(dot(&mut ppu, true), "dot 4: stays on");
6939
6940 // Rendering goes OFF: the same two-dot delay, in the other direction.
6941 assert!(dot(&mut ppu, false), "dot 5: gate still sees rendering on");
6942 assert!(dot(&mut ppu, false), "dot 6: gate still sees rendering on");
6943 assert!(
6944 !dot(&mut ppu, false),
6945 "dot 7: the OFF edge has shifted through"
6946 );
6947 }
6948
6949 #[test]
6950 fn reset_clears_the_rendering_gate_pipeline() {
6951 // `reset` preserves dot/scanline, so gate history that survives it is
6952 // history from before the reset being applied after it. Every stage
6953 // must come back false, or a reset mid-scanline can fire the dot-256
6954 // vertical increment with PPUMASK empty.
6955 let mut ppu = Ppu::new(PpuRegion::Ntsc);
6956 ppu.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
6957 ppu.prev_rendering_enabled = true;
6958 ppu.rendering_enabled_delayed = true;
6959 ppu.rendering_enabled_delayed2 = true;
6960
6961 ppu.reset();
6962
6963 assert!(ppu.mask.is_empty(), "reset clears PPUMASK");
6964 assert!(
6965 !ppu.prev_rendering_enabled,
6966 "reset must clear prev_rendering_enabled"
6967 );
6968 assert!(
6969 !ppu.rendering_enabled_delayed,
6970 "reset must clear rendering-gate stage 1"
6971 );
6972 assert!(
6973 !ppu.rendering_enabled_delayed2,
6974 "reset must clear rendering-gate stage 2 -- otherwise a reset at dot \
6975 254 reaches dot 256 with rendering history from before the reset"
6976 );
6977 }
6978
6979 #[test]
6980 fn odd_frame_dot_skip_is_ntsc_only() {
6981 // The pre-render dot-339 odd-frame skip only fires on NTSC with
6982 // rendering enabled. Drive a rendering-enabled odd pre-render frame in
6983 // each region and confirm only NTSC collapses dot 340.
6984 fn skips(region: PpuRegion) -> bool {
6985 let mut ppu = Ppu::new(region);
6986 // Force an odd frame, rendering on, parked at pre-render dot 339.
6987 ppu.frame = 1;
6988 ppu.mask = PpuMask::SHOW_BG;
6989 ppu.mask_for_skip_check = PpuMask::SHOW_BG;
6990 ppu.scanline = region.prerender_line();
6991 ppu.dot = 339;
6992 ppu.advance_dot();
6993 // A skip lands us at (scanline 0, dot 0); no skip steps to dot 340.
6994 ppu.scanline == 0 && ppu.dot == 0
6995 }
6996 assert!(skips(PpuRegion::Ntsc), "NTSC odd frame skips dot 340");
6997 assert!(!skips(PpuRegion::Pal), "PAL never skips");
6998 assert!(!skips(PpuRegion::Dendy), "Dendy never skips");
6999 }
7000
7001 /// Test bus that owns 8 KiB of CHR-RAM with horizontal mirroring map.
7002 /// CIRAM lives in the PPU; this bus only services CHR + A12.
7003 struct TestBus {
7004 chr: [u8; 0x2000],
7005 a12_count: u32,
7006 last_a12: bool,
7007 }
7008
7009 impl TestBus {
7010 fn new() -> Self {
7011 Self {
7012 chr: [0u8; 0x2000],
7013 a12_count: 0,
7014 last_a12: false,
7015 }
7016 }
7017 }
7018
7019 impl PpuBus for TestBus {
7020 fn ppu_read(&mut self, addr: u16) -> u8 {
7021 if addr < 0x2000 {
7022 self.chr[addr as usize]
7023 } else {
7024 0
7025 }
7026 }
7027 fn ppu_write(&mut self, addr: u16, value: u8) {
7028 if addr < 0x2000 {
7029 self.chr[addr as usize] = value;
7030 }
7031 }
7032 fn notify_a12(&mut self, level: bool) {
7033 if level != self.last_a12 {
7034 self.a12_count += 1;
7035 self.last_a12 = level;
7036 }
7037 }
7038 fn nametable_address(&self, addr: u16) -> u16 {
7039 // Horizontal mirroring: tables 0/1 -> bank 0, 2/3 -> bank 1.
7040 let table = ((addr.wrapping_sub(0x2000)) / 0x0400) & 0x03;
7041 let local = addr & 0x03FF;
7042 let phys = u16::from(table >= 2);
7043 phys * 0x0400 + local
7044 }
7045 }
7046
7047 fn fresh_ppu() -> (Ppu, TestBus) {
7048 let mut ppu = Ppu::new(PpuRegion::Ntsc);
7049 // Drive past the post-reset masking window.
7050 ppu.post_reset_mask_remaining = 0;
7051 (ppu, TestBus::new())
7052 }
7053
7054 /// v3.1.0 (`T-PAL-EMPHASIS`, ACC-01): on the PAL (2C07) and Dendy PPUs
7055 /// PPUMASK bits 5 and 6 swap meaning. `NESdev` "Colour emphasis": "Bit 5
7056 /// emphasizes red on the NTSC PPU, and green on the PAL & Dendy PPUs. Bit 6
7057 /// emphasizes green on the NTSC PPU, and red on the PAL & Dendy PPUs. Bit 7
7058 /// emphasizes blue on the NTSC, PAL, & Dendy PPUs." The emphasis index the
7059 /// renderer and the composite filters receive is the PHYSICAL tint (bit 0
7060 /// red, bit 1 green, bit 2 blue), so on PAL / Dendy it is the mask's bits
7061 /// with 5 and 6 exchanged.
7062 #[test]
7063 fn pal_and_dendy_swap_the_red_and_green_emphasis_bits() {
7064 let cases = [
7065 (PpuMask::EMPHASIZE_RED, 0b001u16, 0b010u16),
7066 (PpuMask::EMPHASIZE_GREEN, 0b010, 0b001),
7067 (PpuMask::EMPHASIZE_BLUE, 0b100, 0b100),
7068 (
7069 PpuMask::EMPHASIZE_RED | PpuMask::EMPHASIZE_BLUE,
7070 0b101,
7071 0b110,
7072 ),
7073 ];
7074 for region in [PpuRegion::Ntsc, PpuRegion::Pal, PpuRegion::Dendy] {
7075 for (mask, ntsc, swapped) in cases {
7076 let mut p = Ppu::new(region);
7077 p.post_reset_mask_remaining = 0;
7078 p.mask = mask; // rendering off: the pixel is the backdrop
7079 p.palette_ram[palette_index(0x3F00)] = 0x21;
7080 p.scanline = 10;
7081 p.dot = 20;
7082 p.emit_pixel();
7083 let got = p.index_framebuffer[10 * 256 + 19];
7084 let want_emph = if region == PpuRegion::Ntsc {
7085 ntsc
7086 } else {
7087 swapped
7088 };
7089 assert_eq!(
7090 got,
7091 (want_emph << 6) | 0x21,
7092 "{region:?}, mask {:#04x}: emphasis index",
7093 mask.bits()
7094 );
7095 let off = (10usize * 256 + 19) * 4;
7096 assert_eq!(
7097 &p.framebuffer[off..off + 4],
7098 &p.rgba_lut[usize::from(got)],
7099 "{region:?}: the RGBA pixel follows the same index"
7100 );
7101 }
7102 }
7103 }
7104
7105 // F1.1 (Fathom accuracy remediation) — palette backdrop-override.
7106 // When rendering is disabled and the VRAM address `v` points into palette
7107 // space ($3F00-$3FFF), the palette's shared address input is driven by `v`,
7108 // so the PPU outputs the color at `v & 0x1F` INSTEAD of the universal
7109 // backdrop ($3F00). This is a display artifact only — palette RAM is never
7110 // mutated, and rendering-enabled output is unchanged. See `NESdev` "PPU
7111 // palettes"; mirrors Mesen2 `NesPpu.cpp` / ares output-stage behavior.
7112 #[test]
7113 fn palette_backdrop_override_when_rendering_disabled() {
7114 let (mut p, _b) = fresh_ppu();
7115 p.mask = PpuMask::empty(); // rendering disabled
7116 p.palette_ram[palette_index(0x3F00)] = 0x0F; // backdrop
7117 p.palette_ram[palette_index(0x3F05)] = 0x16; // override target (red)
7118 p.scanline = 10;
7119 p.dot = 20; // pixel_x = 19 (visible)
7120 let off = (10usize * 256 + 19) * 4;
7121 let red = crate::palette::nes_color_to_rgba(0x16);
7122 let backdrop = crate::palette::nes_color_to_rgba(0x0F);
7123
7124 // v in palette range, rendering off -> output palette[v & 0x1F].
7125 p.v = 0x3F05;
7126 p.emit_pixel();
7127 assert_eq!(&p.framebuffer[off..off + 4], &red, "override -> palette[5]");
7128
7129 // v NOT in palette range, rendering off -> universal backdrop.
7130 p.v = 0x2000;
7131 p.emit_pixel();
7132 assert_eq!(&p.framebuffer[off..off + 4], &backdrop, "non-palette v");
7133
7134 // Rendering ENABLED with transparent BG -> backdrop, never overridden
7135 // (the fetch pipeline owns `v` while rendering).
7136 p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_BG_LEFT;
7137 p.v = 0x3F05;
7138 p.emit_pixel();
7139 assert_eq!(
7140 &p.framebuffer[off..off + 4],
7141 &backdrop,
7142 "enabled -> no override"
7143 );
7144
7145 // $3F10 mirrors to $3F00 (universal backdrop), not a distinct entry.
7146 p.mask = PpuMask::empty();
7147 p.v = 0x3F10;
7148 p.emit_pixel();
7149 assert_eq!(
7150 &p.framebuffer[off..off + 4],
7151 &backdrop,
7152 "$3F10 mirrors backdrop"
7153 );
7154
7155 // The override is display-only: palette RAM is unmodified.
7156 assert_eq!(p.palette_ram[palette_index(0x3F05)], 0x16);
7157 }
7158
7159 // F1.2 (Fathom) — OAM / $2004 quirks. Both behaviors below are already
7160 // implemented and covered by the AccuracyCoin `$2004`/`Sprite0Hit` ROMs;
7161 // this is a FAST regression guard so an edit trips a unit test instead of
7162 // only the ~57s ROM battery. (The `OAMADDR & 0xF8` render-start copy is NOT
7163 // modeled on the DEFAULT revision — Mesen2, ares, and TriCNES all omit it as
7164 // a revision-dependent, oracle-less corner. As of v2.1.7 P5 the related
7165 // OAMADDR `$2003` write-during-render corruption is available as an opt-in
7166 // `PpuRevision::Rp2c02G` model; see `docs/accuracy-ledger.md`.)
7167 #[test]
7168 fn oam_2004_attribute_mask_and_oamaddr_257_320_forcing() {
7169 // (1) $2004 read of a sprite ATTRIBUTE byte (OAM offset & 3 == 2) masks
7170 // bits 4-2 with $E3 (they don't exist in OAM); other bytes are unmasked.
7171 // Read outside the rendering windows so the plain OAM path is taken.
7172 let (mut p, mut b) = fresh_ppu();
7173 p.mask = PpuMask::empty(); // rendering disabled
7174 p.scanline = 250; // vblank -> not a render scanline
7175 p.oam[2] = 0xFF; // attribute byte
7176 p.oam[1] = 0xFF; // tile byte (no mask)
7177 p.oam_addr = 2;
7178 assert_eq!(p.cpu_read_register(4, &mut b), 0xE3, "attr byte $E3-masked");
7179 p.oam_addr = 1;
7180 assert_eq!(
7181 p.cpu_read_register(4, &mut b),
7182 0xFF,
7183 "non-attr byte unmasked"
7184 );
7185
7186 // (2) OAMADDR is forced to 0 across dots 257-320 of a rendered scanline
7187 // (the sprite-tile-load interval), washing away a perturbed value.
7188 let (mut p, mut b) = fresh_ppu();
7189 p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
7190 p.scanline = 10; // visible render line
7191 p.dot = 256;
7192 p.oam_addr = 0x40; // perturbed; nothing but the 257-320 wash zeroes it
7193 for _ in 0..70 {
7194 p.tick(&mut b); // dot 256 -> 326, through the whole window
7195 }
7196 assert_eq!(p.oam_addr, 0, "OAMADDR washed to 0 across dots 257-320");
7197 }
7198
7199 // v2.1.4 F2.3 — optional OAM decay (opt-in, default-OFF). Models Mesen2's
7200 // `ReadSpriteRam`: a row un-refreshed for > OAM_DECAY_CPU_CYCLES CPU cycles
7201 // decays to `((sprAddr & 3) == 2) ? (sprAddr & 0xE3) : sprAddr` on the next
7202 // read. The read path used here is the plain (non-rendering) `$2004` read —
7203 // `mask` empty + a vblank scanline keeps out of the rendering / dot-1-64 /
7204 // dot-257-320 forcing windows.
7205 #[test]
7206 fn oam_decay_disabled_by_default_leaves_oam_untouched() {
7207 let (mut p, mut b) = fresh_ppu();
7208 p.mask = PpuMask::empty();
7209 p.scanline = 250; // vblank — plain OAM read path
7210 // Seed a distinctive value in row 0 (bytes 0..8).
7211 for i in 0..8u8 {
7212 p.oam[i as usize] = 0xAA;
7213 }
7214 // Advance the clock WELL past the decay window with no OAM access.
7215 p.dot_counter = (OAM_DECAY_CPU_CYCLES + 10_000) * 3;
7216 // Default is disabled — a read must return the seeded byte, and OAM must
7217 // be byte-for-byte unchanged (no decay pattern written).
7218 p.oam_addr = 0;
7219 assert!(!p.oam_decay_enabled(), "decay off by default");
7220 assert_eq!(
7221 p.cpu_read_register(4, &mut b),
7222 0xAA,
7223 "no decay when disabled"
7224 );
7225 for i in 0..8usize {
7226 assert_eq!(p.oam[i], 0xAA, "OAM row untouched when decay disabled");
7227 }
7228 }
7229
7230 #[test]
7231 fn oam_decay_enabled_decays_stale_row_to_mesen_pattern() {
7232 let (mut p, mut b) = fresh_ppu();
7233 p.set_oam_decay(true);
7234 p.mask = PpuMask::empty();
7235 p.scanline = 250;
7236 // Seed row 3 (OAM $18..$20) with a value distinct from the decay pattern.
7237 for a in 0x18u8..0x20 {
7238 p.oam[a as usize] = 0x5A;
7239 }
7240 // `set_oam_decay(true)` re-based the timestamps to the then-current cycle
7241 // (0). Advance PAST the window so the row is stale on the next read.
7242 p.dot_counter = (OAM_DECAY_CPU_CYCLES + 1) * 3;
7243 // Read byte $1A (an attribute byte: $1A & 3 == 2) — the whole row decays
7244 // first, then the (now-decayed) byte is returned. Expected decay byte for
7245 // $1A = $1A & 0xE3 = $02; but `$2004` additionally $E3-masks an attr byte
7246 // on the way out ($02 & $E3 == $02), so the observed value is $02.
7247 p.oam_addr = 0x1A;
7248 let v = p.cpu_read_register(4, &mut b);
7249 assert_eq!(v, 0x1A & 0xE3, "attribute byte decays to sprAddr & 0xE3");
7250 // The full row now holds the canonical pattern.
7251 for a in 0x18u8..0x20 {
7252 let expect = if a & 0x03 == 0x02 { a & 0xE3 } else { a };
7253 assert_eq!(p.oam[a as usize], expect, "row byte ${a:02X} decayed");
7254 }
7255 }
7256
7257 #[test]
7258 fn oam_decay_access_within_window_refreshes_row() {
7259 let (mut p, mut b) = fresh_ppu();
7260 p.set_oam_decay(true);
7261 p.mask = PpuMask::empty();
7262 p.scanline = 250;
7263 for a in 0x18u8..0x20 {
7264 p.oam[a as usize] = 0x5A;
7265 }
7266 // Touch the row just before the window closes (elapsed == threshold →
7267 // still a refresh, not a decay), which re-stamps the timestamp.
7268 p.dot_counter = OAM_DECAY_CPU_CYCLES * 3;
7269 p.oam_addr = 0x18;
7270 assert_eq!(
7271 p.cpu_read_register(4, &mut b),
7272 0x5A,
7273 "in-window read: no decay"
7274 );
7275 // Advance another (threshold) cycles from the refresh point — still within
7276 // the window relative to the refreshed timestamp, so no decay.
7277 p.dot_counter += OAM_DECAY_CPU_CYCLES * 3;
7278 p.oam_addr = 0x19;
7279 assert_eq!(
7280 p.cpu_read_register(4, &mut b),
7281 0x5A,
7282 "refresh kept row alive"
7283 );
7284 for a in 0x18u8..0x20 {
7285 assert_eq!(p.oam[a as usize], 0x5A, "row still holds seeded data");
7286 }
7287 }
7288
7289 #[test]
7290 fn oam_decay_is_pal_disabled() {
7291 // PAL's frequent refresh cadence masks decay, so the model never acts
7292 // there even when enabled (matches Mesen2). Same stale-row setup as the
7293 // NTSC decay test, but on PAL the read must NOT decay.
7294 let mut p = Ppu::new(PpuRegion::Pal);
7295 p.post_reset_mask_remaining = 0;
7296 let mut b = TestBus::new();
7297 p.set_oam_decay(true);
7298 p.mask = PpuMask::empty();
7299 p.scanline = 250;
7300 for a in 0x18u8..0x20 {
7301 p.oam[a as usize] = 0x5A;
7302 }
7303 p.dot_counter = (OAM_DECAY_CPU_CYCLES + 1) * 3;
7304 p.oam_addr = 0x18;
7305 assert_eq!(p.cpu_read_register(4, &mut b), 0x5A, "PAL: decay disabled");
7306 for a in 0x18u8..0x20 {
7307 assert_eq!(p.oam[a as usize], 0x5A, "PAL row untouched");
7308 }
7309 }
7310
7311 #[test]
7312 fn oam_decay_write_refreshes_row() {
7313 let (mut p, mut b) = fresh_ppu();
7314 p.set_oam_decay(true);
7315 p.mask = PpuMask::empty();
7316 p.scanline = 250;
7317 for a in 0x18u8..0x20 {
7318 p.oam[a as usize] = 0x5A;
7319 }
7320 // A `$2004` write of row 3 well past the window still refreshes it, so a
7321 // subsequent in-window read of that row does not decay.
7322 p.dot_counter = (OAM_DECAY_CPU_CYCLES + 5_000) * 3;
7323 p.oam_addr = 0x18;
7324 p.cpu_write_register(4, 0x33, &mut b); // writes $18, refreshes row 3
7325 // Read $19 (same row) a short time later — within the window of the write.
7326 p.dot_counter += 10 * 3;
7327 p.oam_addr = 0x19;
7328 assert_eq!(p.cpu_read_register(4, &mut b), 0x5A, "write kept row alive");
7329 }
7330
7331 // v2.1.7 P5 — PPU revision + power-up palette (opt-in, default-off).
7332
7333 #[test]
7334 fn revision_defaults_to_rp2c02h_no_corruption() {
7335 let (p, _b) = fresh_ppu();
7336 assert_eq!(p.revision(), PpuRevision::Rp2c02H, "default revision");
7337 assert!(
7338 !p.revision().models_oamaddr_corruption(),
7339 "default revision models no OAMADDR corruption"
7340 );
7341 }
7342
7343 #[test]
7344 fn default_revision_2003_write_during_render_does_not_corrupt() {
7345 // On the default revision a $2003 write mid-render must NOT arm any OAM
7346 // corruption — the byte-identity guarantee.
7347 let (mut p, mut b) = fresh_ppu();
7348 p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
7349 p.scanline = 10; // visible render line
7350 // Distinct row-0 vs row-1 so a spurious copy would be observable.
7351 for i in 0..8u8 {
7352 p.oam[i as usize] = 0x11;
7353 p.oam[8 + i as usize] = 0x22;
7354 }
7355 p.cpu_write_register(3, 0x08, &mut b); // OAMADDR = row 1
7356 assert!(
7357 !p.oam_corruption_pending,
7358 "default revision: no corruption armed"
7359 );
7360 }
7361
7362 #[test]
7363 fn rp2c02g_2003_write_during_render_arms_and_corrupts_row() {
7364 // On the earlier `Rp2c02G` revision a $2003 write while rendering is
7365 // active arms the row-copy corruption; committing it copies row 0 over
7366 // the targeted row (index = value >> 3).
7367 let (mut p, mut b) = fresh_ppu();
7368 p.set_revision(PpuRevision::Rp2c02G);
7369 p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE;
7370 p.scanline = 10; // visible render line
7371 for i in 0..8u8 {
7372 p.oam[i as usize] = 0x11; // row 0
7373 p.oam[8 + i as usize] = 0x22; // row 1 (target)
7374 }
7375 p.cpu_write_register(3, 0x08, &mut b); // OAMADDR = 0x08 → row index 1
7376 assert!(p.oam_corruption_pending, "Rp2c02G: corruption armed");
7377 assert_eq!(p.oam_corruption_index, 1, "targets row 1");
7378 // Commit and verify row 1 now mirrors row 0.
7379 p.process_oam_corruption();
7380 for i in 0..8u8 {
7381 assert_eq!(
7382 p.oam[8 + i as usize],
7383 0x11,
7384 "row 1 byte {i} corrupted from row 0"
7385 );
7386 }
7387 }
7388
7389 #[test]
7390 fn rp2c02g_2003_write_outside_render_does_not_corrupt() {
7391 // Even on the corrupting revision, a $2003 write with rendering disabled
7392 // (or in vblank) must NOT arm corruption — the glitch is render-gated.
7393 let (mut p, mut b) = fresh_ppu();
7394 p.set_revision(PpuRevision::Rp2c02G);
7395 p.mask = PpuMask::empty(); // rendering disabled
7396 p.scanline = 250; // vblank
7397 p.cpu_write_register(3, 0x08, &mut b);
7398 assert!(
7399 !p.oam_corruption_pending,
7400 "Rp2c02G but no rendering: no corruption"
7401 );
7402 }
7403
7404 #[test]
7405 fn power_up_palette_defaults_zeroed() {
7406 let (p, _b) = fresh_ppu();
7407 assert_eq!(p.power_up_palette(), PaletteInit::Zeroed, "default palette");
7408 assert_eq!(
7409 p.palette_ram, [0u8; 32],
7410 "default power-up palette all-zero"
7411 );
7412 }
7413
7414 #[test]
7415 fn power_up_palette_blargg_applies_masked_pattern() {
7416 let (mut p, _b) = fresh_ppu();
7417 p.apply_power_up_palette(PaletteInit::Blargg);
7418 assert_eq!(p.power_up_palette(), PaletteInit::Blargg);
7419 // Byte 0 = 0x09, an attr-index that survives the 6-bit mask untouched.
7420 assert_eq!(p.palette_ram[0], 0x09, "Blargg byte 0");
7421 // Every cell must be 6-bit masked (matching a `$2007` write path).
7422 for (i, &b) in p.palette_ram.iter().enumerate() {
7423 assert_eq!(b, BLARGG_POWER_UP_PALETTE[i] & 0x3F, "cell {i} masked");
7424 }
7425 // Re-applying Zeroed restores the byte-identical default state.
7426 p.apply_power_up_palette(PaletteInit::Zeroed);
7427 assert_eq!(p.palette_ram, [0u8; 32], "re-zeroed");
7428 }
7429
7430 // F1.3 (Fathom) — PPU open-bus refresh map. The Blargg `ppu_open_bus` table
7431 // is: a read DRIVES (and refreshes) some bits and passes others through from
7432 // the decay latch. $2000-$2003/$2005/$2006 = all decay; $2004 + $2007
7433 // (non-palette) = all driven; $2002 = `---D DDDD` (bits 7-5 driven); $2007
7434 // palette = `DD-- ----` (bits 7-6 decay). The $2002 low-5 case is covered by
7435 // `ppustatus_*` above; this locks the $2007-palette and write-only cases.
7436 /// v2.9.5 (a review of #575): on the odd-frame-deferred pixel a drawing
7437 /// sprite emits its FIRST column at X=0 whatever its own X, so the HD-pack
7438 /// tile source must record column 0 there. With X=10 the original
7439 /// `(pixel_x - x) & 7` recorded column 6.
7440 #[cfg(feature = "hd-pack")]
7441 #[test]
7442 fn hd_source_records_column_zero_on_the_deferred_pixel() {
7443 let (mut p, _b) = fresh_ppu();
7444 p.mask = PpuMask::SHOW_SPRITE | PpuMask::SHOW_SPRITE_LEFT;
7445 p.scanline = 0;
7446 p.dot = 1; // pixel 0
7447 p.spr_count = 1;
7448 p.spr_x[0] = 10;
7449 p.hd_spr_x[0] = 10;
7450 p.spr_halted[0] = true; // the drawing state the skip leaves it in
7451 p.spr_shift_lo[0] = 0x80; // an opaque first column
7452 p.spr_attr[0] = 0;
7453 p.spr_rearm_deferred = true;
7454 p.emit_pixel();
7455 let rec = p.hd_tile_source()[0];
7456 assert!(rec.is_sprite, "the deferred pixel must be the sprite's");
7457 assert_eq!(rec.offset_x, 0, "the first column is drawn at X=0");
7458 assert!(!p.spr_rearm_deferred, "released after pixel 0");
7459 }
7460
7461 /// An OAM DMA byte is a `$2004` write, and "writing any value to any PPU
7462 /// port ... will fill this latch" (`nesdev_wiki/PPU_registers`, the
7463 /// `_io_db` latch). So `$2002`'s low five bits read the last DMA byte.
7464 /// v2.9.5, the sibling's `oracle-vs-documentation.md` 3.1c: the `MiSTer`
7465 /// core already did this. The oracle did not, and every `AccuracyCoin`
7466 /// entry reports the same result either way, `Open Bus` included.
7467 #[test]
7468 fn oam_dma_byte_fills_the_io_latch() {
7469 let (mut p, mut b) = fresh_ppu();
7470 p.open_bus = 0x00;
7471 p.status = PpuStatus::empty();
7472 p.oam_dma_write(0xFF);
7473 assert_eq!(p.cpu_read_register(2, &mut b) & 0x1F, 0x1F);
7474 }
7475
7476 #[test]
7477 fn open_bus_refresh_map_2007_palette_and_write_only() {
7478 // Reading a WRITE-ONLY register drives no bits -> the full decay latch.
7479 let (mut p, mut b) = fresh_ppu();
7480 p.open_bus = 0xA5;
7481 assert_eq!(
7482 p.cpu_read_register(0, &mut b),
7483 0xA5,
7484 "$2000 read = pure open bus"
7485 );
7486
7487 // $2007 PALETTE read drives bits 5-0 (palette) and passes bits 7-6 from
7488 // open bus (Blargg map: palette = `DD-- ----`).
7489 let (mut p, mut b) = fresh_ppu();
7490 p.open_bus = 0xFF; // bits 7-6 set
7491 p.mask = PpuMask::empty(); // no render-window $FF path
7492 p.v = 0x3F00;
7493 p.palette_ram[palette_index(0x3F00)] = 0x15;
7494 assert_eq!(
7495 p.cpu_read_register(7, &mut b),
7496 0x15 | 0xC0,
7497 "$2007 palette: bits 5-0 palette, 7-6 open bus"
7498 );
7499 }
7500
7501 #[test]
7502 fn ppustatus_read_clears_vbl_and_w() {
7503 let (mut p, mut b) = fresh_ppu();
7504 p.status.insert(PpuStatus::VBLANK);
7505 p.w = true;
7506 let v = p.cpu_read_register(2, &mut b);
7507 assert!(v & 0x80 != 0, "VBL should have been set on read");
7508 assert!(!p.status.contains(PpuStatus::VBLANK));
7509 assert!(!p.w);
7510 }
7511
7512 #[test]
7513 fn default_ppu_uses_composite_palette_no_2c05() {
7514 let (p, _b) = fresh_ppu();
7515 assert_eq!(p.active_palette, crate::palette::PpuPalette::Composite2C02);
7516 assert!(!p.is_2c05);
7517 // map_register is the identity on a non-2C05 PPU.
7518 for r in 0u8..8 {
7519 assert_eq!(p.map_register(r), r);
7520 }
7521 }
7522
7523 #[test]
7524 fn c2c05_swaps_2000_and_2001() {
7525 let (mut p, mut b) = fresh_ppu();
7526 p.set_palette(crate::palette::PpuPalette::Rgb2C05, true, 0x3D);
7527 // A write to $2000 (reg 0) on a 2C05 sets MASK; a write to $2001 sets
7528 // CTRL. Use a distinct, register-valid value for each.
7529 // PPUMASK bit 3 = SHOW_BG. PPUCTRL bit 7 = NMI_ENABLE.
7530 p.cpu_write_register(0, 0b0000_1000, &mut b); // -> MASK SHOW_BG
7531 assert!(p.mask.contains(PpuMask::SHOW_BG), "$2000 write set MASK");
7532 assert!(p.ctrl.is_empty(), "$2000 write did NOT touch CTRL");
7533
7534 let (mut p2, mut b2) = fresh_ppu();
7535 p2.set_palette(crate::palette::PpuPalette::Rgb2C05, true, 0x3D);
7536 p2.cpu_write_register(1, 0b1000_0000, &mut b2); // -> CTRL NMI_ENABLE
7537 assert!(
7538 p2.ctrl.contains(PpuCtrl::NMI_ENABLE),
7539 "$2001 write set CTRL on a 2C05"
7540 );
7541 assert!(p2.mask.is_empty(), "$2001 write did NOT touch MASK");
7542 }
7543
7544 #[test]
7545 fn c2c05_2002_returns_identifier_in_low_bits() {
7546 let (mut p, mut b) = fresh_ppu();
7547 p.set_palette(crate::palette::PpuPalette::Rgb2C05, true, 0x3D);
7548 // Set the VBL flag so the high bits are deterministic.
7549 p.status.insert(PpuStatus::VBLANK);
7550 let v = p.cpu_read_register(2, &mut b);
7551 // 2C05-02 id = $3D; low 5 bits => $3D & $1F = $1D.
7552 assert_eq!(v & 0x1F, 0x3D & 0x1F);
7553 assert!(v & 0x80 != 0, "VBL still reported in bit 7");
7554 }
7555
7556 #[test]
7557 fn non_2c05_2002_keeps_open_bus_low_bits() {
7558 // Without is_2c05, the low 5 bits remain open-bus (byte-identical to
7559 // the legacy path).
7560 let (mut p, mut b) = fresh_ppu();
7561 p.cpu_write_register(3, 0x1F, &mut b); // load open bus with $1F
7562 let v = p.cpu_read_register(2, &mut b);
7563 assert_eq!(v & 0x1F, 0x1F);
7564 }
7565
7566 #[test]
7567 fn ppustatus_low_5_bits_are_open_bus() {
7568 let (mut p, mut b) = fresh_ppu();
7569 // Touch the open-bus latch via a $2003 write.
7570 p.cpu_write_register(3, 0xAB, &mut b);
7571 p.status.insert(PpuStatus::VBLANK);
7572 let v = p.cpu_read_register(2, &mut b);
7573 // Bits 7-5 from status (only VBL set), bits 4-0 from open-bus (0x0B).
7574 assert_eq!(v & 0xE0, 0x80);
7575 assert_eq!(v & 0x1F, 0xAB & 0x1F);
7576 }
7577
7578 #[test]
7579 fn ppustatus_read_preserves_low_5_bits_of_open_bus_latch() {
7580 // Reading $2002 only refreshes the upper 3 bits of the open-bus
7581 // latch (the bits sourced from PPUSTATUS); the lower 5 bits must
7582 // retain their previous value. Required by the `open_bus_read_test`
7583 // sub-routine of `cpu_dummy_writes_ppumem.nes` (Bisqwit), which
7584 // performs `lda $2002; eor $2000` and expects the result to be 0
7585 // after AND-masking with 0x1F.
7586 let (mut p, mut b) = fresh_ppu();
7587 // Seed the open-bus latch via a $2003 write; pick a value with low
7588 // bits set so the bug-fix is observable.
7589 p.cpu_write_register(3, 0xAB, &mut b);
7590 p.status.insert(PpuStatus::VBLANK);
7591 // Read $2002 — should expose status high bits + latch low 5 bits.
7592 let v = p.cpu_read_register(2, &mut b);
7593 assert_eq!(v, 0x80 | (0xAB & 0x1F));
7594 // Now read $2000 (write-only): should return the refreshed latch
7595 // = (status & 0xE0) | (old_latch & 0x1F) — i.e., the same value.
7596 let after = p.cpu_read_register(0, &mut b);
7597 assert_eq!(
7598 after, v,
7599 "$2002 read must refresh only the high 3 bits of open-bus; \
7600 the low 5 bits must survive into subsequent reads of \
7601 write-only ports"
7602 );
7603 }
7604
7605 #[test]
7606 fn ppudata_buffered_read_returns_previous_byte() {
7607 let (mut p, mut b) = fresh_ppu();
7608 // CIRAM lives in the PPU now.
7609 p.ciram[0] = 0xAB;
7610 p.ciram[1] = 0xCD;
7611 // Set v to $2000.
7612 p.cpu_write_register(6, 0x20, &mut b);
7613 p.cpu_write_register(6, 0x00, &mut b);
7614 // First read: returns buffer (0), refills from $2000.
7615 let r1 = p.cpu_read_register(7, &mut b);
7616 assert_eq!(r1, 0);
7617 // Second read: returns refill (0xAB), refills with next byte.
7618 let r2 = p.cpu_read_register(7, &mut b);
7619 assert_eq!(r2, 0xAB);
7620 let r3 = p.cpu_read_register(7, &mut b);
7621 assert_eq!(r3, 0xCD);
7622 }
7623
7624 #[test]
7625 fn ppudata_palette_read_bypasses_buffer() {
7626 let (mut p, mut b) = fresh_ppu();
7627 p.palette_ram[0] = 0x12;
7628 // Stash a different value in the underlying nametable mirror so we
7629 // see the buffer get the underlying value, not the palette byte.
7630 p.ciram[0] = 0xCC;
7631 // Set v to $3F00.
7632 p.cpu_write_register(6, 0x3F, &mut b);
7633 p.cpu_write_register(6, 0x00, &mut b);
7634 let r = p.cpu_read_register(7, &mut b);
7635 // High 2 bits open-bus. Low 6 bits: 0x12.
7636 assert_eq!(r & 0x3F, 0x12);
7637 // Buffer should now contain underlying nametable mirror at $2F00
7638 // (= $3F00 & $2FFF), via horizontal mirroring tables 2/3 -> bank 1.
7639 }
7640
7641 #[test]
7642 fn ppudata_increment_1_or_32() {
7643 let (mut p, mut b) = fresh_ppu();
7644 p.cpu_write_register(6, 0x21, &mut b);
7645 p.cpu_write_register(6, 0x00, &mut b);
7646 // Increment by 1 default.
7647 p.cpu_read_register(7, &mut b);
7648 assert_eq!(p.v & 0x7FFF, 0x2101);
7649 // Switch to increment 32.
7650 p.cpu_write_register(0, PpuCtrl::VRAM_INCREMENT_32.bits(), &mut b);
7651 p.cpu_read_register(7, &mut b);
7652 assert_eq!(p.v & 0x7FFF, 0x2121);
7653 }
7654
7655 #[test]
7656 fn ppuctrl_post_reset_mask_window_blocks_writes() {
7657 let mut p = Ppu::new(PpuRegion::Ntsc);
7658 // Don't override post_reset_mask_remaining — it's the documented
7659 // count.
7660 let mut b = TestBus::new();
7661 p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7662 assert!(
7663 !p.ctrl.contains(PpuCtrl::NMI_ENABLE),
7664 "PPUCTRL write must be ignored during post-reset window"
7665 );
7666 // Drive past the window.
7667 for _ in 0..30_000 {
7668 p.on_cpu_cycle();
7669 }
7670 p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7671 assert!(p.ctrl.contains(PpuCtrl::NMI_ENABLE));
7672 }
7673
7674 /// v2.9.8 — `end_warmup` closes the post-reset window at once: the four
7675 /// masked registers (`$2000`/`$2001`/`$2005`/`$2006`, `NESdev` "PPU power up
7676 /// state") accept the very next write, with no CPU cycles elapsed. This is
7677 /// the PPU half of the opt-in Famicom console model.
7678 #[test]
7679 fn end_warmup_lets_masked_registers_write_immediately() {
7680 let mut p = Ppu::new(PpuRegion::Ntsc);
7681 let mut b = TestBus::new();
7682 assert_eq!(p.warmup_cycles_remaining(), 29_658);
7683 p.end_warmup();
7684 assert_eq!(p.warmup_cycles_remaining(), 0);
7685 p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7686 assert!(p.ctrl.contains(PpuCtrl::NMI_ENABLE), "$2000 accepted");
7687 // Greyscale only: a visible PPUMASK change that leaves rendering off,
7688 // so the `$2006` pair below copies `t -> v` at once.
7689 p.cpu_write_register(1, 0x01, &mut b);
7690 assert_eq!(p.mask.bits(), 0x01, "$2001 accepted");
7691 p.cpu_write_register(6, 0x21, &mut b);
7692 p.cpu_write_register(6, 0x08, &mut b);
7693 assert_eq!(p.v & 0x3FFF, 0x2108, "$2006 pair accepted");
7694 p.cpu_write_register(5, 0x08, &mut b);
7695 assert!(p.w, "$2005 toggled the write latch");
7696 }
7697
7698 #[test]
7699 fn ppuctrl_nmi_enable_during_vbl_asserts_nmi_immediately() {
7700 let (mut p, mut b) = fresh_ppu();
7701 p.status.insert(PpuStatus::VBLANK);
7702 // NMI not yet enabled => line low.
7703 assert!(!p.nmi_line);
7704 p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7705 assert!(p.nmi_line);
7706 }
7707
7708 #[test]
7709 fn ppuscroll_two_writes_load_t_and_x() {
7710 let (mut p, mut b) = fresh_ppu();
7711 p.cpu_write_register(5, 0b1010_1011, &mut b); // X = 0xAB
7712 // t bits 4-0 = X[7:3] = 0b10101 = 0x15. x = X[2:0] = 0b011 = 0x03.
7713 assert_eq!(p.t & 0x001F, 0x15);
7714 assert_eq!(p.x, 0x03);
7715 assert!(p.w);
7716 p.cpu_write_register(5, 0b0101_1100, &mut b); // Y = 0x5C
7717 // t bits 14-12 = Y[2:0] = 0b100, t bits 9-5 = Y[7:3] = 0b01011.
7718 assert_eq!((p.t >> 12) & 0x07, 0x04);
7719 assert_eq!((p.t >> 5) & 0x1F, 0x0B);
7720 assert!(!p.w);
7721 }
7722
7723 #[test]
7724 fn ppuaddr_two_writes_copy_t_to_v() {
7725 let (mut p, mut b) = fresh_ppu();
7726 p.cpu_write_register(6, 0x3F, &mut b); // high
7727 // After first write t bits 13-8 = 0x3F & 0x3F; bit 14 cleared.
7728 assert_eq!((p.t >> 8) & 0x7F, 0x3F);
7729 assert!(p.w);
7730 p.cpu_write_register(6, 0x10, &mut b); // low; copy t to v
7731 assert_eq!(p.v, 0x3F10);
7732 assert!(!p.w);
7733 }
7734
7735 #[test]
7736 fn vbl_set_and_nmi_at_scanline_241_dot_1() {
7737 let (mut p, mut b) = fresh_ppu();
7738 p.cpu_write_register(0, PpuCtrl::NMI_ENABLE.bits(), &mut b);
7739 // Tick until scanline 241 dot 1.
7740 // Starting at pre-render dot 0 (after construction we set scanline
7741 // = prerender_line, dot = 0). Tick advances first. We need to
7742 // reach scanline 241 dot 1. Simplest: just tick enough.
7743 let mut saw_nmi = false;
7744 for _ in 0..(341 * 263) {
7745 p.tick(&mut b);
7746 if p.nmi_line {
7747 saw_nmi = true;
7748 break;
7749 }
7750 }
7751 assert!(saw_nmi, "NMI must assert during VBlank");
7752 assert!(p.status.contains(PpuStatus::VBLANK));
7753 }
7754
7755 #[test]
7756 fn frame_complete_latch_fires_once_per_frame() {
7757 let (mut p, mut b) = fresh_ppu();
7758 // Tick a full frame's worth.
7759 let mut frames_seen = 0;
7760 for _ in 0..(341 * 262 * 2) {
7761 p.tick(&mut b);
7762 if p.take_frame_complete() {
7763 frames_seen += 1;
7764 }
7765 }
7766 assert!(frames_seen >= 2);
7767 }
7768
7769 #[test]
7770 fn index_framebuffer_mirrors_rgba_output() {
7771 // T-110-A1: the parallel palette-index framebuffer must be a faithful
7772 // index-space mirror of the RGBA framebuffer — for every emitted pixel,
7773 // `rgba_lut[index] == framebuffer[pixel]`. This is the contract that
7774 // makes the index buffer a safe, determinism-neutral output: it carries
7775 // exactly the LUT index used to produce the displayed RGBA.
7776 let (mut p, mut b) = fresh_ppu();
7777 // Enable background rendering so the full visible area is emitted.
7778 p.cpu_write_register(1, 0x08, &mut b); // PPUMASK: show background
7779 // Run two full frames so every visible pixel has been written.
7780 for _ in 0..(341 * 262 * 2) {
7781 p.tick(&mut b);
7782 }
7783 let fb = p.framebuffer();
7784 let idx = p.index_framebuffer();
7785 assert_eq!(idx.len(), FRAMEBUFFER_PIXELS);
7786 for (i, &lut_idx) in idx.iter().enumerate() {
7787 assert!((lut_idx as usize) < 512, "index in range at pixel {i}");
7788 let expected = p.rgba_lut[lut_idx as usize];
7789 assert_eq!(
7790 &fb[i * 4..i * 4 + 4],
7791 &expected,
7792 "pixel {i}: index {lut_idx} must reproduce the RGBA output"
7793 );
7794 }
7795 }
7796
7797 #[test]
7798 fn ntsc_phase_in_range_and_crawls() {
7799 // The per-frame NTSC phase must stay in 0..=2 (NTSC) and visit more than
7800 // one value across frames (the dot-crawl the filter reproduces).
7801 let (mut p, mut b) = fresh_ppu();
7802 p.cpu_write_register(1, 0x08, &mut b); // rendering on (odd-frame skip active)
7803 let mut seen = [false; 3];
7804 for _ in 0..(341 * 262 * 8) {
7805 p.tick(&mut b);
7806 if p.take_frame_complete() {
7807 let ph = p.ntsc_phase();
7808 assert!(ph <= 2, "NTSC phase {ph} must be 0..=2");
7809 seen[ph as usize] = true;
7810 }
7811 }
7812 let distinct = seen.iter().filter(|&&s| s).count();
7813 assert!(
7814 distinct >= 2,
7815 "phase must crawl across frames (saw {distinct})"
7816 );
7817 }
7818
7819 #[test]
7820 fn palette_mirrors_3f10_alias_3f00() {
7821 let (mut p, mut b) = fresh_ppu();
7822 p.cpu_write_register(6, 0x3F, &mut b);
7823 p.cpu_write_register(6, 0x10, &mut b); // v = $3F10
7824 p.cpu_write_register(7, 0x21, &mut b); // write palette
7825 // The mirror should land at index 0 (= $3F00).
7826 assert_eq!(p.palette_ram[0], 0x21);
7827 assert_eq!(p.palette_ram[0x10], 0); // not actually written
7828 }
7829
7830 #[test]
7831 fn oamdata_write_increments_oamaddr() {
7832 let (mut p, mut b) = fresh_ppu();
7833 p.oam_addr = 0x40;
7834 p.cpu_write_register(4, 0xCC, &mut b);
7835 assert_eq!(p.oam[0x40], 0xCC);
7836 assert_eq!(p.oam_addr, 0x41);
7837 }
7838
7839 /// v2.9.7 — the A12 stream the PPU reports is the HARDWARE stream, not
7840 /// MMC3's filtered view of it.
7841 ///
7842 /// Standard layout (BG at `$0000`, sprites at `$1000`), rendering on. In
7843 /// each of the eight 8-dot sprite-fetch slots (dots 257-320) the PPU reads
7844 /// two garbage nametable bytes (`$2xxx`, A12 low) and then the sprite's two
7845 /// pattern bytes (`$1xxx`, A12 high). So A12 rises **eight times per
7846 /// rendered line**: 240 visible lines plus the pre-render line give
7847 /// `8 * 241 = 1928` per NTSC frame.
7848 ///
7849 /// MMC3 sees one of those per line because its filter ignores a rise
7850 /// unless A12 was low for about three CPU cycles (nine dots). The first
7851 /// slot's rise follows the long low of the background fetches; every later
7852 /// slot's rise follows a four-dot low. The same filter, applied here to
7853 /// the raw stream, must therefore still count exactly 241.
7854 ///
7855 /// Until v2.9.7 the PPU never reported the garbage nametable reads, so it
7856 /// emitted MMC3's view (241 rises) as if it were the stream, and this test
7857 /// pinned that. Boards that count raw edges were starved eightfold:
7858 /// Acclaim's MC-ACC (a divide-by-8 on every edge), whose six local test
7859 /// games lost their status bars, and mapper 91 submapper 0, whose page
7860 /// says it counts "64 unfiltered rises of PPU A12".
7861 #[test]
7862 fn a12_reports_the_hardware_stream_and_an_mmc3_filter_still_sees_241() {
7863 struct CountingBus {
7864 chr: [u8; 0x2000],
7865 dot: u64,
7866 last_a12: bool,
7867 low_since: u64,
7868 raw_rises: u32,
7869 filtered_rises: u32,
7870 }
7871 impl PpuBus for CountingBus {
7872 fn ppu_read(&mut self, addr: u16) -> u8 {
7873 if addr < 0x2000 {
7874 self.chr[addr as usize]
7875 } else {
7876 0
7877 }
7878 }
7879 fn ppu_write(&mut self, addr: u16, value: u8) {
7880 if addr < 0x2000 {
7881 self.chr[addr as usize] = value;
7882 }
7883 }
7884 fn notify_a12(&mut self, level: bool) {
7885 if level == self.last_a12 {
7886 return;
7887 }
7888 if level {
7889 self.raw_rises += 1;
7890 // MMC3-style filter, in dots: low for at least 9 dots
7891 // (three CPU cycles on NTSC).
7892 if self.dot.saturating_sub(self.low_since) >= 9 {
7893 self.filtered_rises += 1;
7894 }
7895 } else {
7896 self.low_since = self.dot;
7897 }
7898 self.last_a12 = level;
7899 }
7900 fn nametable_address(&self, addr: u16) -> u16 {
7901 let table = ((addr.wrapping_sub(0x2000)) / 0x0400) & 0x03;
7902 let local = addr & 0x03FF;
7903 let phys = u16::from(table >= 2);
7904 phys * 0x0400 + local
7905 }
7906 }
7907 let mut p = Ppu::new(PpuRegion::Ntsc);
7908 p.post_reset_mask_remaining = 0;
7909 let mut b = CountingBus {
7910 chr: [0u8; 0x2000],
7911 dot: 0,
7912 last_a12: false,
7913 low_since: 0,
7914 raw_rises: 0,
7915 filtered_rises: 0,
7916 };
7917 p.cpu_write_register(0, PpuCtrl::SPRITE_PATTERN_HIGH.bits(), &mut b);
7918 p.cpu_write_register(1, (PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE).bits(), &mut b);
7919 while !(p.scanline() == 0 && p.dot() == 0) {
7920 p.tick(&mut b);
7921 b.dot += 1;
7922 }
7923 b.raw_rises = 0;
7924 b.filtered_rises = 0;
7925 let start_frame = p.frame();
7926 while p.frame() == start_frame {
7927 p.tick(&mut b);
7928 b.dot += 1;
7929 }
7930 assert_eq!(b.raw_rises, 8 * 241, "eight A12 rises per rendered line");
7931 assert_eq!(
7932 b.filtered_rises, 241,
7933 "an MMC3-style filter still sees one rise per rendered line"
7934 );
7935 }
7936
7937 /// T-MMC3-BG-A12: with the background at `$1000` and sprites at `$0000`,
7938 /// the `NESdev` MMC3 page says the counter "should decrement on PPU cycle
7939 /// 324 of the previous scanline" -- one dot before the pattern-low
7940 /// fetch's ALE dot (325), the same convention the sprite path follows
7941 /// (its rises land at 260, the page's figure for the other arrangement).
7942 /// Until this change the background reported A12 at its READ dot, two
7943 /// dots later (326, and 6 for the line's own first tile), which put a
7944 /// rise caught at the first dot of a CPU cycle one cycle late and failed
7945 /// blargg `4-scanline_timing` at sub-test 9.
7946 #[test]
7947 fn background_a12_rises_at_the_mmc3_pages_dot_324() {
7948 struct RiseBus {
7949 chr: [u8; 0x2000],
7950 last_a12: bool,
7951 rises: alloc::vec::Vec<(i16, u16)>,
7952 }
7953 impl PpuBus for RiseBus {
7954 fn ppu_read(&mut self, addr: u16) -> u8 {
7955 if addr < 0x2000 {
7956 self.chr[addr as usize]
7957 } else {
7958 0
7959 }
7960 }
7961 fn ppu_write(&mut self, _addr: u16, _value: u8) {}
7962 fn notify_a12(&mut self, level: bool) {
7963 if level && !self.last_a12 {
7964 self.rises.push((0, 0));
7965 }
7966 self.last_a12 = level;
7967 }
7968 fn nametable_address(&self, addr: u16) -> u16 {
7969 addr & 0x07FF
7970 }
7971 }
7972 let mut p = Ppu::new(PpuRegion::Ntsc);
7973 p.post_reset_mask_remaining = 0;
7974 let mut b = RiseBus {
7975 chr: [0u8; 0x2000],
7976 last_a12: false,
7977 rises: alloc::vec::Vec::new(),
7978 };
7979 p.cpu_write_register(0, PpuCtrl::BG_PATTERN_HIGH.bits(), &mut b);
7980 p.cpu_write_register(1, (PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE).bits(), &mut b);
7981 // Run to scanline 9, then record lines 9-10.
7982 while p.scanline() != 9 {
7983 p.tick(&mut b);
7984 }
7985 b.rises.clear();
7986 // `tick` advances to the next dot and then processes it, so a rise is
7987 // stamped with the position AFTER the tick that produced it.
7988 while p.scanline() != 11 {
7989 let before = b.rises.len();
7990 p.tick(&mut b);
7991 if b.rises.len() > before {
7992 *b.rises.last_mut().unwrap() = (p.scanline(), p.dot());
7993 }
7994 }
7995 // Scanline 10's first tile: the prefetch on line 9, and the line's
7996 // own fetches (first visible-tile group starts at dot 1).
7997 let prefetch: alloc::vec::Vec<u16> = b
7998 .rises
7999 .iter()
8000 .filter(|r| r.0 == 9 && r.1 > 320)
8001 .map(|r| r.1)
8002 .collect();
8003 assert_eq!(
8004 prefetch,
8005 [324, 332],
8006 "the next line's two prefetched tiles rise at 324 and 332"
8007 );
8008 // A visible line's dot 0 drives "the same CHR address that is later
8009 // used to fetch the low background tile byte starting at dot 5"
8010 // (NESdev PPU rendering, "Cycle 0"), so with the background at
8011 // `$1000` A12 is already high there.
8012 let first = b.rises.iter().find(|r| r.0 == 10).map(|r| r.1);
8013 assert_eq!(
8014 first,
8015 Some(0),
8016 "a visible line's dot 0 drives the BG CHR address"
8017 );
8018 }
8019
8020 /// T-MMC3-BG-A12: scanline 0's dot 0 drives the background CHR address
8021 /// (A12 high with the background at `$1000`) on an EVEN frame, and not on
8022 /// an odd frame whose pre-render skip "replac[es] the idle tick at the
8023 /// beginning of the first visible scanline with the last tick of the last
8024 /// dummy nametable fetch" (`NESdev` PPU rendering). This test is the
8025 /// exception's only pin: blargg `4-scanline_timing` synchronises with
8026 /// `sync_vbl_even`, so no ROM in the corpus reaches the odd-frame case,
8027 /// and removing the exception leaves both 4-scanline ROMs passing
8028 /// (measured 2026-10-05).
8029 #[test]
8030 fn scanline_0_dot_0_drives_bg_chr_only_when_the_skip_did_not_replace_it() {
8031 struct LevelBus {
8032 chr: [u8; 0x2000],
8033 level: bool,
8034 }
8035 impl PpuBus for LevelBus {
8036 fn ppu_read(&mut self, addr: u16) -> u8 {
8037 if addr < 0x2000 {
8038 self.chr[addr as usize]
8039 } else {
8040 0
8041 }
8042 }
8043 fn ppu_write(&mut self, _addr: u16, _value: u8) {}
8044 fn notify_a12(&mut self, level: bool) {
8045 self.level = level;
8046 }
8047 fn nametable_address(&self, addr: u16) -> u16 {
8048 addr & 0x07FF
8049 }
8050 }
8051 let mut p = Ppu::new(PpuRegion::Ntsc);
8052 p.post_reset_mask_remaining = 0;
8053 let mut b = LevelBus {
8054 chr: [0u8; 0x2000],
8055 level: false,
8056 };
8057 p.cpu_write_register(0, PpuCtrl::BG_PATTERN_HIGH.bits(), &mut b);
8058 p.cpu_write_register(1, (PpuMask::SHOW_BG | PpuMask::SHOW_SPRITE).bits(), &mut b);
8059 let (mut skipped, mut kept) = (0, 0);
8060 for _ in 0..4 {
8061 // Run to the end of the pre-render line, remembering the parity
8062 // of the frame being completed.
8063 while !(p.scanline() == p.region.prerender_line() && p.dot() == 338) {
8064 p.tick(&mut b);
8065 }
8066 let odd = p.frame() & 1 == 1;
8067 // Step until scanline 0's dot 0 has been processed.
8068 while !(p.scanline() == 0 && p.dot() == 0) {
8069 p.tick(&mut b);
8070 }
8071 if odd {
8072 assert!(
8073 !b.level,
8074 "an odd frame's skip replaces dot 0: A12 stays low"
8075 );
8076 skipped += 1;
8077 } else {
8078 assert!(b.level, "an even frame's dot 0 drives the BG CHR address");
8079 kept += 1;
8080 }
8081 }
8082 assert!(skipped > 0 && kept > 0, "both parities were exercised");
8083 }
8084
8085 #[test]
8086 fn a12_transitions_notify_bus() {
8087 let (mut p, mut b) = fresh_ppu();
8088 // Set v to $1234 (A12 high), then $0234 (A12 low) — two transitions.
8089 p.cpu_write_register(6, 0x12, &mut b);
8090 p.cpu_write_register(6, 0x34, &mut b);
8091 assert_eq!(b.a12_count, 1);
8092 p.cpu_write_register(6, 0x02, &mut b);
8093 p.cpu_write_register(6, 0x34, &mut b);
8094 assert_eq!(b.a12_count, 2);
8095 }
8096
8097 // -------------------------------------------------------------------
8098 // T-23-002: sprite-evaluation FSM with buggy n+m overflow increment.
8099 // -------------------------------------------------------------------
8100
8101 /// Regression: 8 in-range sprites must populate secondary OAM and
8102 /// leave `spr_count == 8` without setting the `SPRITE_OVERFLOW` flag,
8103 /// PROVIDED the diagonal-read scan over the remaining 56 sprites
8104 /// never lands on an in-range byte. To pin that condition we fill
8105 /// the entire off-screen OAM region with 0xF0, so every byte the
8106 /// buggy `n+m` walk could land on reads as y=240 (out of range).
8107 #[test]
8108 fn sprite_eval_8_sprites_no_overflow() {
8109 let (mut p, _b) = fresh_ppu();
8110 p.scanline = 0;
8111 // 8 in-range sprites with non-zero, non-conflicting byte values
8112 // that don't read as "in-range y" if the diagonal walk hits them.
8113 for i in 0..8 {
8114 let base = i * 4;
8115 p.oam[base] = 0; // y = 0 (in range)
8116 p.oam[base + 1] = 0xF0; // tile (also out of range if read as y)
8117 p.oam[base + 2] = 0xF0;
8118 p.oam[base + 3] = 0xF0;
8119 }
8120 // Sprites 8..63: every byte = 0xF0 so diagonal read finds nothing.
8121 for i in 8..64 {
8122 for j in 0..4 {
8123 p.oam[i * 4 + j] = 0xF0;
8124 }
8125 }
8126 run_per_dot_fsm(&mut p);
8127 assert_eq!(p.spr_count, 8, "exactly 8 in-range sprites must fill");
8128 assert!(
8129 !p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8130 "8 sprites + all-off-screen-remainder is not overflow"
8131 );
8132 }
8133
8134 /// The headline case: 9 in-range sprites must set `SPRITE_OVERFLOW`.
8135 /// On real hardware the buggy `n+m` increment reads the wrong byte
8136 /// of sprite #9, but here sprite #9 is in-range and its y-byte
8137 /// (which the diagonal walk reads first at n=9, m=0 if found==8)
8138 /// is in-range, so the flag fires.
8139 #[test]
8140 fn sprite_eval_9_sprites_sets_overflow() {
8141 let (mut p, _b) = fresh_ppu();
8142 p.scanline = 0;
8143 for i in 0..9 {
8144 let base = i * 4;
8145 p.oam[base] = 0; // y = 0 (in range)
8146 p.oam[base + 1] = 0xF0; // tile (out of range as y)
8147 p.oam[base + 2] = 0xF0;
8148 p.oam[base + 3] = 0xF0;
8149 }
8150 for i in 9..64 {
8151 for j in 0..4 {
8152 p.oam[i * 4 + j] = 0xF0;
8153 }
8154 }
8155 run_per_dot_fsm(&mut p);
8156 assert_eq!(p.spr_count, 8, "secondary OAM holds first 8 only");
8157 assert!(
8158 p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8159 "9 in-range sprites must set overflow"
8160 );
8161 }
8162
8163 /// Empty OAM: no in-range sprites, no overflow.
8164 #[test]
8165 fn sprite_eval_empty_oam_no_overflow() {
8166 let (mut p, _b) = fresh_ppu();
8167 p.scanline = 0;
8168 // Every byte off-screen, so the eval pass never finds anything
8169 // and never enters overflow-detection mode.
8170 for byte in &mut p.oam {
8171 *byte = 0xF0;
8172 }
8173 run_per_dot_fsm(&mut p);
8174 assert_eq!(p.spr_count, 0);
8175 assert!(!p.status.contains(PpuStatus::SPRITE_OVERFLOW));
8176 }
8177
8178 /// The buggy `n+m` increment: when 8 sprites have been found, the
8179 /// overflow-detection FSM reads `OAM[n*4+m].y` and increments BOTH
8180 /// `n` and `m` together on each iteration. If sprite #9 is OUT of
8181 /// range but sprite #10's *non-y byte* (which the bug reads as a
8182 /// y-coordinate) happens to be in-range, the overflow flag will
8183 /// fire — that's the documented hardware quirk, not a bug in our
8184 /// FSM.
8185 ///
8186 /// Construct a case where:
8187 /// - Sprites 0..7 are in-range (fill secondary OAM, found = 8).
8188 /// - Sprite 8's y is far off-screen (y = 0xF0, normal y-read would
8189 /// say not-in-range).
8190 /// - Sprite 9's TILE byte (byte index 1, which the buggy m=1 read
8191 /// when n=9 lands on) is set to a value that, interpreted as y,
8192 /// would put the sprite on the next scanline.
8193 ///
8194 /// With the buggy FSM the overflow flag fires because the diagonal
8195 /// read finds sprite 9's tile byte (= 0) as a "y" that maps to a
8196 /// row in-range for an 8-tall sprite. A correct (non-buggy) FSM
8197 /// reading sprite #8's y first would NOT fire because sprite 8 is
8198 /// out of range.
8199 ///
8200 /// This test pins the buggy behavior; flipping it to non-buggy
8201 /// would change the assertion direction.
8202 #[test]
8203 fn sprite_eval_buggy_n_plus_m_finds_diagonal_overflow() {
8204 let (mut p, _b) = fresh_ppu();
8205 p.scanline = 0;
8206 // Start with the entire OAM off-screen.
8207 for byte in &mut p.oam {
8208 *byte = 0xF0;
8209 }
8210 // Sprites 0..7 in-range with all non-y bytes off-screen.
8211 for i in 0..8 {
8212 let base = i * 4;
8213 p.oam[base] = 0; // y = 0 (in range)
8214 // bytes 1,2,3 keep the 0xF0 fill so a stray read
8215 // doesn't mis-fire the diagonal test.
8216 }
8217 // Sprite 8 y is 0xF0 (from the bulk fill) — out of range.
8218 // Sprite 9 tile byte (OAM[9*4+1]) is the second diagonal read
8219 // target (after sprite 8's y). Setting it to 0 (= in-range y)
8220 // forces the buggy FSM to fire overflow on the SECOND iteration
8221 // of the inner loop.
8222 p.oam[9 * 4 + 1] = 0;
8223 run_per_dot_fsm(&mut p);
8224 assert_eq!(p.spr_count, 8);
8225 assert!(
8226 p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8227 "buggy n+m increment must find the diagonal-read overflow at sprite 9 byte 1"
8228 );
8229 }
8230
8231 // -------------------------------------------------------------------
8232 // Sprite-eval FSM regression corpus. Originally introduced as the
8233 // parallel-implementation firewall gating the B8 swap from single-
8234 // shot to per-dot FSM. The single-shot collapse was removed in B8c;
8235 // these tests are now the regression net pinning the FSM's observable
8236 // output against a straight-line reference implementation
8237 // (`reference_eval`).
8238 //
8239 // The corpus targets:
8240 // - Empty OAM (no in-range)
8241 // - Exactly 8 in-range (no overflow)
8242 // - 9+ in-range (clean overflow)
8243 // - Diagonal-read scenarios (sprite N out-of-range, sprite (N+k)'s
8244 // non-y byte in-range)
8245 // - 8x8 + 8x16 sprite heights
8246 // - Boundary scanlines (0, 1, 239, prerender)
8247 //
8248 // Random fuzz + structured edge cases combined give 1013 cases.
8249 // -------------------------------------------------------------------
8250
8251 /// Tiny xorshift PRNG so the test is hermetic (no `rand` dep).
8252 struct XorShift(u64);
8253 impl XorShift {
8254 const fn new(seed: u64) -> Self {
8255 Self(if seed == 0 {
8256 0xDEAD_BEEF_CAFE_BABE
8257 } else {
8258 seed
8259 })
8260 }
8261 const fn next_u64(&mut self) -> u64 {
8262 let mut x = self.0;
8263 x ^= x << 13;
8264 x ^= x >> 7;
8265 x ^= x << 17;
8266 self.0 = x;
8267 x
8268 }
8269 fn next_u8(&mut self) -> u8 {
8270 (self.next_u64() & 0xFF) as u8
8271 }
8272 }
8273
8274 /// Snapshot of the observable post-dot-256 state for equivalence
8275 /// comparison.
8276 #[derive(Debug, Clone, PartialEq, Eq)]
8277 struct EvalObservable {
8278 secondary_oam: [u8; 32],
8279 spr_count: u8,
8280 spr_zero_in_line: bool,
8281 overflow: bool,
8282 }
8283
8284 fn observe(p: &Ppu) -> EvalObservable {
8285 EvalObservable {
8286 secondary_oam: p.secondary_oam,
8287 spr_count: p.spr_count,
8288 spr_zero_in_line: p.spr_zero_in_line,
8289 overflow: p.status.contains(PpuStatus::SPRITE_OVERFLOW),
8290 }
8291 }
8292
8293 /// Build a fresh PPU and seed `oam`, `scanline`, and `ctrl` from the
8294 /// given parameters.
8295 fn build_case(oam: &[u8; 256], scanline: i16, ctrl: PpuCtrl) -> Ppu {
8296 let mut p = Ppu::new(PpuRegion::Ntsc);
8297 p.post_reset_mask_remaining = 0;
8298 p.oam.copy_from_slice(oam);
8299 p.scanline = scanline;
8300 p.ctrl = ctrl;
8301 // Reset the overflow flag so we can observe per-case sets.
8302 p.status.remove(PpuStatus::SPRITE_OVERFLOW);
8303 // Pre-fill secondary OAM with a poison value so the per-dot FSM's
8304 // clear phase is observable (single-shot also starts by writing
8305 // $FF into all 32 bytes, so the final state must match).
8306 p.secondary_oam = [0xAA; 32];
8307 p.spr_count = 0;
8308 p.spr_zero_in_line = false;
8309 p
8310 }
8311
8312 /// Drive the per-dot FSM through dots 0..=256 on `p`.
8313 fn run_per_dot_fsm(p: &mut Ppu) {
8314 for dot in 0..=256u16 {
8315 p.dot = dot;
8316 p.tick_sprite_eval_per_dot();
8317 }
8318 }
8319
8320 /// Run one case through the FSM and assert observable matches the
8321 /// expected pinned state. The expected state is built by computing
8322 /// the result in a non-buggy reference implementation (the
8323 /// `reference_eval` below).
8324 fn assert_case_matches(label: &str, oam: &[u8; 256], scanline: i16, ctrl: PpuCtrl) {
8325 let expected = reference_eval(oam, scanline, ctrl);
8326
8327 let mut pf = build_case(oam, scanline, ctrl);
8328 run_per_dot_fsm(&mut pf);
8329 let actual = observe(&pf);
8330
8331 assert_eq!(
8332 expected,
8333 actual,
8334 "FSM mismatch for case `{label}` \
8335 (scanline={scanline}, 8x16={}, sprite_zero_y={:#04x})",
8336 ctrl.contains(PpuCtrl::SPRITE_SIZE_16),
8337 oam[0],
8338 );
8339 }
8340
8341 /// Reference implementation: a straight-line sprite-eval emulation
8342 /// matching the 2C02's behavior, used as the golden expected output
8343 /// for the FSM regression corpus. Originally the FSM was validated
8344 /// against the old single-shot collapse via the 1013-case equivalence
8345 /// harness (B8a); after B8c removed the single-shot, this stand-alone
8346 /// reference plays the same role.
8347 fn reference_eval(oam: &[u8; 256], scanline: i16, ctrl: PpuCtrl) -> EvalObservable {
8348 // Y-test convention: see `tick_sprite_eval_per_dot` docstring.
8349 // Pre-render uses -1 (always-fail), visible uses the current
8350 // scanline; sprite Y=N renders on scanlines N+1..=N+h.
8351 let next_line: i16 = if scanline == PpuRegion::Ntsc.prerender_line() {
8352 -1
8353 } else {
8354 scanline
8355 };
8356 let sprite_height: i16 = if ctrl.contains(PpuCtrl::SPRITE_SIZE_16) {
8357 16
8358 } else {
8359 8
8360 };
8361
8362 let mut secondary_oam = [0xFFu8; 32];
8363 let mut found = 0u8;
8364 let mut spr_zero_in_line = false;
8365 let mut overflow = false;
8366
8367 let mut n_idx = 0usize;
8368 while n_idx < 64 {
8369 let base = n_idx * 4;
8370 let y = oam[base] as i16;
8371 let row = next_line - y;
8372 if row >= 0 && row < sprite_height {
8373 let sec_base = (found as usize) * 4;
8374 secondary_oam[sec_base] = oam[base];
8375 secondary_oam[sec_base + 1] = oam[base + 1];
8376 secondary_oam[sec_base + 2] = oam[base + 2];
8377 secondary_oam[sec_base + 3] = oam[base + 3];
8378 if n_idx == 0 {
8379 spr_zero_in_line = true;
8380 }
8381 found += 1;
8382 if found == 8 {
8383 n_idx += 1;
8384 let mut m = 0u8;
8385 while n_idx < 64 {
8386 let nb = n_idx * 4 + (m as usize);
8387 let by = oam[nb] as i16;
8388 let brow = next_line - by;
8389 if brow >= 0 && brow < sprite_height {
8390 overflow = true;
8391 break;
8392 }
8393 m = (m + 1) & 0x03;
8394 n_idx += 1;
8395 }
8396 break;
8397 }
8398 }
8399 n_idx += 1;
8400 }
8401
8402 EvalObservable {
8403 secondary_oam,
8404 spr_count: found,
8405 spr_zero_in_line,
8406 overflow,
8407 }
8408 }
8409
8410 #[test]
8411 fn sprite_fsm_equivalence_edge_cases() {
8412 // 1: empty OAM (all 0xFF y) -> no found, no overflow.
8413 let mut oam = [0xFFu8; 256];
8414 assert_case_matches("empty_oam_y_ff", &oam, 0, PpuCtrl::empty());
8415
8416 // 2: every byte 0xF0 (out of range) -> no found, no overflow.
8417 oam = [0xF0u8; 256];
8418 assert_case_matches("empty_oam_y_f0", &oam, 0, PpuCtrl::empty());
8419
8420 // 3: 8 in-range sprites, all other bytes 0xF0 -> 8 found, no
8421 // overflow.
8422 oam = [0xF0u8; 256];
8423 for i in 0..8 {
8424 oam[i * 4] = 0;
8425 }
8426 assert_case_matches("8_in_range", &oam, 0, PpuCtrl::empty());
8427
8428 // 4: 9 in-range sprites -> overflow set.
8429 oam = [0xF0u8; 256];
8430 for i in 0..9 {
8431 oam[i * 4] = 0;
8432 }
8433 assert_case_matches("9_in_range", &oam, 0, PpuCtrl::empty());
8434
8435 // 5: 8 in-range + diagonal-read overflow (sprite 9 byte 1 = 0
8436 // forces buggy n+m to fire).
8437 oam = [0xF0u8; 256];
8438 for i in 0..8 {
8439 oam[i * 4] = 0;
8440 }
8441 oam[9 * 4 + 1] = 0;
8442 assert_case_matches("diagonal_overflow", &oam, 0, PpuCtrl::empty());
8443
8444 // 6: 8x16 sprite mode.
8445 oam = [0xF0u8; 256];
8446 for i in 0..3 {
8447 oam[i * 4] = 0;
8448 }
8449 assert_case_matches("8x16_mode", &oam, 0, PpuCtrl::SPRITE_SIZE_16);
8450
8451 // 7: pre-render line (evaluates for scanline 0).
8452 oam = [0xF0u8; 256];
8453 for i in 0..5 {
8454 oam[i * 4] = 0;
8455 }
8456 let prerender = PpuRegion::Ntsc.prerender_line();
8457 assert_case_matches("prerender_line", &oam, prerender, PpuCtrl::empty());
8458
8459 // 8: last visible scanline.
8460 oam = [0xF0u8; 256];
8461 for i in 0..2 {
8462 oam[i * 4] = 239;
8463 }
8464 assert_case_matches("scanline_239", &oam, 238, PpuCtrl::empty());
8465
8466 // 9: sprite zero NOT in range -> spr_zero_in_line must stay false.
8467 oam = [0xF0u8; 256];
8468 oam[0] = 0xF0; // sprite 0 out of range
8469 for i in 1..3 {
8470 oam[i * 4] = 0;
8471 }
8472 assert_case_matches("zero_out_of_range", &oam, 0, PpuCtrl::empty());
8473
8474 // 10: sprite zero in range but not first -> still must be true
8475 // because sprite 0 is at OAM index 0.
8476 oam = [0xF0u8; 256];
8477 oam[0] = 0; // sprite 0 in range
8478 for i in 5..10 {
8479 oam[i * 4] = 0;
8480 }
8481 assert_case_matches("zero_in_range_plus_others", &oam, 0, PpuCtrl::empty());
8482
8483 // 11: exactly 1 in-range at the last possible sprite (sprite 63).
8484 oam = [0xF0u8; 256];
8485 oam[63 * 4] = 0;
8486 assert_case_matches("only_sprite_63", &oam, 0, PpuCtrl::empty());
8487
8488 // 12: 8 in-range scattered among the 64 entries.
8489 oam = [0xF0u8; 256];
8490 for (slot, &n) in [0u8, 5, 11, 18, 27, 35, 44, 55].iter().enumerate() {
8491 let _ = slot;
8492 oam[(n as usize) * 4] = 0;
8493 }
8494 assert_case_matches("8_scattered", &oam, 0, PpuCtrl::empty());
8495
8496 // 13: all 64 sprites in range -> 8 found + overflow.
8497 oam = [0u8; 256];
8498 for i in 0..64 {
8499 oam[i * 4] = 0; // y = 0
8500 oam[i * 4 + 1] = 0xAB;
8501 oam[i * 4 + 2] = 0xCD;
8502 oam[i * 4 + 3] = 0xEF;
8503 }
8504 assert_case_matches("all_64_in_range", &oam, 0, PpuCtrl::empty());
8505 }
8506
8507 #[test]
8508 fn sprite_fsm_equivalence_randomized_corpus() {
8509 // 1000 fully-random cases + the 13 edge cases above = 1013 total
8510 // regression checks. Each invocation runs the FSM on a random
8511 // OAM/scanline/ctrl seed and asserts observable equality with
8512 // the straight-line reference implementation.
8513 const N: usize = 1000;
8514 let mut rng = XorShift::new(0x1234_5678_9ABC_DEF0);
8515
8516 for case in 0..N {
8517 let mut oam = [0u8; 256];
8518 for b in &mut oam {
8519 *b = rng.next_u8();
8520 }
8521 // Choose scanline from {0..=239, prerender=261}. Use a bias
8522 // toward 0..=239 since that's the realistic case.
8523 let r = rng.next_u64();
8524 let scanline: i16 = if r.trailing_zeros() >= 5 {
8525 PpuRegion::Ntsc.prerender_line()
8526 } else {
8527 ((r >> 8) & 0xFF) as i16 % 240
8528 };
8529 // 8x16 mode in 1/4 of cases.
8530 let ctrl = if rng.next_u64().trailing_zeros() >= 2 {
8531 PpuCtrl::SPRITE_SIZE_16
8532 } else {
8533 PpuCtrl::empty()
8534 };
8535
8536 let expected = reference_eval(&oam, scanline, ctrl);
8537
8538 let mut pf = build_case(&oam, scanline, ctrl);
8539 run_per_dot_fsm(&mut pf);
8540 let actual = observe(&pf);
8541
8542 assert_eq!(
8543 expected,
8544 actual,
8545 "FSM regressed against reference at case #{case} \
8546 (scanline={scanline}, 8x16={}, oam[0]={:#04x})",
8547 ctrl.contains(PpuCtrl::SPRITE_SIZE_16),
8548 oam[0],
8549 );
8550 }
8551 }
8552
8553 /// Cascade A reproducer V3: mimics `AccuracyCoin`'s
8554 /// `VerifySpriteZeroHits` step 2 (the version that EXPECTS a hit).
8555 /// Sprite 0 at Y=5 X=8 tile $C0. BG tile $C0 at nametable $2C21
8556 /// (NT 3 col 1 row 1). v = $2C00.
8557 ///
8558 /// Tile $C0 has a SINGLE opaque pixel at (col=0, row=0). With v=$2C00,
8559 /// BG tile at NT 3 position $21 displays at screen pixels (8, 8).
8560 /// Sprite at (Y=5, X=8) tile $C0 draws at scanline 6 (per nesdev:
8561 /// sprite occupies scanlines Y+1..Y+8). Sprite tile $C0's only opaque
8562 /// pixel is (col 0, row 0) → screen (8, 6).
8563 ///
8564 /// Sprite (8, 6) vs BG (8, 8) — NO geometric overlap. The test asserts
8565 /// a hit IS expected here, which is impossible without sprite Y
8566 /// semantics being different from what nesdev documents. This unit
8567 /// test makes the discrepancy concrete so it can be investigated
8568 /// against Mesen2 or other reference emulators.
8569 #[test]
8570 fn cascade_a_verify_sprite_zero_hits_step2() {
8571 let (mut p, mut b) = fresh_ppu();
8572 // Pin the PPU to (prerender, dot=0) so this diagnostic harness runs
8573 // through exactly one frame starting from the prerender boundary.
8574 // Required because Ppu::new() now starts at (prerender, dot=340)
8575 // per Session-13 Option B (close the +344-dot offset vs Mesen2);
8576 // without this reset the test's "advance one frame" loop would begin
8577 // mid-prerender and the sprite-zero-hit window would shift relative
8578 // to the BG-pipeline cycle-9 reload point this test was designed to
8579 // characterise (see docs/audit/cascade-a-investigation-2026-05-19.md
8580 // and docs/audit/session-13-cpu-boot-fix-2026-05-21.md).
8581 p.dot = 0;
8582 let tile_c0_base = 0xC0 * 16;
8583 // Tile $C0: only the (col 0, row 0) pixel is opaque (lo=$80 hi=$80).
8584 b.chr[tile_c0_base] = 0x80;
8585 b.chr[tile_c0_base + 8] = 0x80;
8586 // Tile $24: fully transparent (all-zero bytes already).
8587 // Fill NT 3 (bank 1 of CIRAM, horizontal mirroring) with $24, then
8588 // write $C0 at position $21.
8589 for i in 0..0x400 {
8590 p.ciram[0x400 + i] = 0x24;
8591 }
8592 p.ciram[0x400 + 0x021] = 0xC0;
8593 // OAM page mimics OAM DMA from a $FF-cleared page + sprite 0 init.
8594 for i in 0..256 {
8595 p.oam[i] = 0xFF;
8596 }
8597 p.oam[0] = 0x05; // Y = 5 (step 2)
8598 p.oam[1] = 0xC0; // CHR
8599 p.oam[2] = 0x03; // ATT
8600 p.oam[3] = 0x08; // X = 8
8601 // v = $2C00 (NT 3 top-left).
8602 p.v = 0x2C00;
8603 p.t = 0x2C00;
8604 // PPUCTRL = 0 (both pattern tables at $0000).
8605 p.ctrl = PpuCtrl::empty();
8606 // Enable rendering.
8607 let mask = PpuMask::SHOW_BG
8608 | PpuMask::SHOW_SPRITE
8609 | PpuMask::SHOW_BG_LEFT
8610 | PpuMask::SHOW_SPRITE_LEFT;
8611 p.mask = mask;
8612 p.mask_skip_pipe1 = mask;
8613 p.mask_for_skip_check = mask;
8614 p.status = PpuStatus::empty();
8615 // Advance ~1 full frame to allow sprite-zero hit to fire if it should.
8616 for _ in 0..(262 * 341) {
8617 p.tick(&mut b);
8618 }
8619 let hit = p.status.contains(PpuStatus::SPRITE_ZERO_HIT);
8620 // POST-FIX EXPECTATION: with the cycle-9 reload + post-emit shift
8621 // BG-pipeline correction landed (see
8622 // `docs/audit/cascade-a-investigation-2026-05-19.md`), tile $C0's
8623 // single opaque BG pixel lands at screen column 8 (PPU dot 9 of
8624 // scanline 6), exactly overlapping the sprite-zero opaque pixel
8625 // at (8, 6) → SPRITE-ZERO HIT must fire.
8626 //
8627 // The test ROM's geometry: sprite Y=5 X=8 tile $C0 has its only
8628 // opaque pixel at sprite-local (col 0, row 0) → screen (8, 6).
8629 // BG tile $C0 at NT 3 position $21 with v=$2C00 (fine Y=2,
8630 // coarse Y=0) renders at scanline 6, screen column 8, with its
8631 // only opaque pixel matching → overlap → hit.
8632 assert!(
8633 hit,
8634 "BG-pipeline fix regression: sprite-zero hit must fire for \
8635 VerifySpriteZeroHits step 2 (BG opaque at (8,6) overlaps \
8636 sprite-zero opaque at (8,6)) — see \
8637 docs/audit/cascade-a-investigation-2026-05-19.md."
8638 );
8639 }
8640
8641 /// Cascade A reproducer V2: start in VBL, enable rendering via the
8642 /// CPU-visible `$2001` write (with the 2-PPU-clock pipeline delay), do
8643 /// OAM DMA via the CPU-visible `$2003 + $2004` writes, then advance
8644 /// past pre-render → scanline 0 → scanline 1. More faithful to the
8645 /// real ROM execution path than the V1 reproducer.
8646 #[test]
8647 fn cascade_a_sprite_zero_hit_y0_x8_via_register_writes() {
8648 let (mut p, mut b) = fresh_ppu();
8649 // Load tile $FC into pattern table 0 fully-opaque.
8650 let tile_fc_base = 0xFC * 16;
8651 for row in 0..8 {
8652 b.chr[tile_fc_base + row] = 0xFF;
8653 b.chr[tile_fc_base + 8 + row] = 0x00;
8654 }
8655 // Write nametable $2001 = $FC via $2006 + $2007 (CPU-visible path).
8656 p.cpu_write_register(6, 0x20, &mut b); // hi
8657 p.cpu_write_register(6, 0x01, &mut b); // lo (v = $2001)
8658 p.cpu_write_register(7, 0xFC, &mut b);
8659 // Reset scroll: v = $2000 via $2006 + $2006.
8660 p.cpu_write_register(6, 0x20, &mut b);
8661 p.cpu_write_register(6, 0x00, &mut b);
8662 // Mimic the ROM's OAM page: ClearPage2 fills with $FF, then
8663 // InitializeSpriteZero writes sprite 0. So OAM[0..4] is the sprite,
8664 // OAM[4..256] is $FF (Y=$FF -> off-screen).
8665 for i in 0..256 {
8666 p.oam[i] = 0xFF;
8667 }
8668 // OAM DMA: write sprite 0 via $2003 (OAMADDR) + $2004 (OAMDATA).
8669 p.cpu_write_register(3, 0x00, &mut b); // OAMADDR = 0
8670 p.cpu_write_register(4, 0x00, &mut b); // sprite 0 Y = 0
8671 p.cpu_write_register(4, 0xFC, &mut b); // sprite 0 CHR = $FC
8672 p.cpu_write_register(4, 0x00, &mut b); // sprite 0 ATT = 0
8673 p.cpu_write_register(4, 0x08, &mut b); // sprite 0 X = 8
8674 // Advance to scanline 241 dot 1 (VBL start) — matches the ROM
8675 // post-WaitForVBlank position.
8676 while !(p.scanline == 241 && p.dot == 1) {
8677 p.tick(&mut b);
8678 }
8679 // Enable rendering via $2001 write (BG + SPR + show-left).
8680 let mask = (PpuMask::SHOW_BG
8681 | PpuMask::SHOW_SPRITE
8682 | PpuMask::SHOW_BG_LEFT
8683 | PpuMask::SHOW_SPRITE_LEFT)
8684 .bits();
8685 p.cpu_write_register(1, mask, &mut b);
8686 // PPUSTATUS may have VBL set; clear sprite-zero-hit start clean.
8687 p.status.remove(PpuStatus::SPRITE_ZERO_HIT);
8688 // Now advance through ~30 scanlines (rest of VBL + pre-render +
8689 // visible 0-9), matching what Clockslide_3000 covers in the ROM.
8690 for _ in 0..(30 * 341) {
8691 p.tick(&mut b);
8692 }
8693 assert!(
8694 p.status.contains(PpuStatus::SPRITE_ZERO_HIT),
8695 "Expected sprite-zero hit set after 30 scanlines past VBL. \
8696 Actual status=0x{:02X}, scanline={}, dot={}, \
8697 spr_count={}, spr_zero_in_line={}, \
8698 spr_x[0]={}, spr_shift_lo[0]=0x{:02X}, spr_shift_hi[0]=0x{:02X}, \
8699 mask=0x{:02X}, ctrl=0x{:02X}",
8700 p.status.bits(),
8701 p.scanline,
8702 p.dot,
8703 p.spr_count,
8704 p.spr_zero_in_line,
8705 p.spr_x[0],
8706 p.spr_shift_lo[0],
8707 p.spr_shift_hi[0],
8708 p.mask.bits(),
8709 p.ctrl.bits(),
8710 );
8711 }
8712
8713 /// Cascade A reproducer: the exact `AccuracyCoin TEST_Sprite0Hit_Behavior`
8714 /// sub-test 1 scenario, constructed directly without going through the
8715 /// CPU/test-ROM.
8716 ///
8717 /// Setup (matches `AccuracyCoin.asm:PREP_SpriteZeroHit` + the test's
8718 /// pre-state):
8719 ///
8720 /// - Sprite 0: `Y=$00, CHR=$FC, ATT=$00, X=$08`.
8721 /// - BG nametable: `vram[$2001] = $FC` (tile $FC at col=1, row=0).
8722 /// - CHR pattern table 0, tile $FC, all 8 rows: `lo=$FF / hi=$00`
8723 /// (fully opaque pixels of palette colour 1).
8724 /// - `PPUMASK = $1E` (BG + SPR + `BG_LEFT` + grayscale; the actual
8725 /// `PPUMASK_COPY` value the diagnostic probe in
8726 /// `crates/rustynes-test-harness/src/accuracy_coin.rs` captures at frame
8727 /// 3393 — see `docs/audit/accuracycoin-readme-analysis-2026-05-17.md`
8728 /// §"Addendum (2026-05-19, session 5)").
8729 /// - `PPUCTRL = $00` (BG and sprite pattern tables both at `$0000`).
8730 /// - `v = $2000` (top-left of nametable 0).
8731 ///
8732 /// **Expected**: sprite zero hit (PPUSTATUS bit 6) is set by the end
8733 /// of scanline 1 — sprite pixel (8..15, 1) overlaps BG pixel (8..15,
8734 /// 1) and both are opaque.
8735 ///
8736 /// **Current (2026-05-19, pre-fix)**: this test FAILS. The
8737 /// diagnostic probe shows PPUSTATUS bit 6 = 0 in the live battery
8738 /// (full ROM run). This unit test is the isolated reproducer.
8739 #[test]
8740 fn cascade_a_sprite_zero_hit_y0_x8_tile_fc_overlap() {
8741 let (mut p, mut b) = fresh_ppu();
8742 // 1. Load tile $FC into pattern table 0 with fully-opaque pixels.
8743 let tile_fc_base = 0xFC * 16;
8744 for row in 0..8 {
8745 b.chr[tile_fc_base + row] = 0xFF; // lo plane (palette bit 0)
8746 b.chr[tile_fc_base + 8 + row] = 0x00; // hi plane (palette bit 1)
8747 }
8748 // 2. Write tile $FC into nametable position $2001 (col=1, row=0).
8749 // CIRAM bank 0 directly (horizontal mirroring: $2000-$23FF -> ciram[0..0x400]).
8750 p.ciram[0x001] = 0xFC;
8751 // 3. Sprite 0: Y=$00, CHR=$FC, ATT=$00, X=$08.
8752 p.oam[0] = 0x00;
8753 p.oam[1] = 0xFC;
8754 p.oam[2] = 0x00;
8755 p.oam[3] = 0x08;
8756 // 4. PPUMASK = SHOW_BG | SHOW_SPRITE | SHOW_BG_LEFT | grayscale.
8757 let mask_bits = PpuMask::SHOW_BG
8758 | PpuMask::SHOW_SPRITE
8759 | PpuMask::SHOW_BG_LEFT
8760 | PpuMask::SHOW_SPRITE_LEFT;
8761 p.mask = mask_bits;
8762 // Pipeline the mask through the two skip-check stages so the
8763 // rendering-enabled signal is stable immediately.
8764 p.mask_skip_pipe1 = mask_bits;
8765 p.mask_for_skip_check = mask_bits;
8766 // 5. PPUCTRL = 0 (BG and sprite both at pattern table 0).
8767 p.ctrl = PpuCtrl::empty();
8768 // 6. v = $2000 (top-left of nametable 0).
8769 p.v = 0x2000;
8770 // Make sure sprite-zero-hit and VBL start clean.
8771 p.status = PpuStatus::empty();
8772 // Pre-render starts; advance ~3 full scanlines so we cross
8773 // pre-render → scanline 0 → scanline 1 → scanline 2. By the end
8774 // of scanline 1, the sprite-zero hit should be set.
8775 // Frame is 262*341 dots. We need at least scanlines 261..=2 = 4
8776 // scanlines = 4*341 = 1364 dots. Use 1500 for safety.
8777 for _ in 0..1500 {
8778 p.tick(&mut b);
8779 }
8780 assert!(
8781 p.status.contains(PpuStatus::SPRITE_ZERO_HIT),
8782 "Expected sprite-zero hit (PPUSTATUS bit 6) to be set after \
8783 scanline 1 with sprite 0 at (Y=0, X=8) tile $FC overlapping \
8784 BG nametable[$2001]=$FC (both fully opaque). \
8785 Actual status=0x{:02X}, scanline={}, dot={}, \
8786 spr_count={}, spr_zero_in_line={}, \
8787 spr_x[0]={}, spr_shift_lo[0]=0x{:02X}, spr_shift_hi[0]=0x{:02X}",
8788 p.status.bits(),
8789 p.scanline,
8790 p.dot,
8791 p.spr_count,
8792 p.spr_zero_in_line,
8793 p.spr_x[0],
8794 p.spr_shift_lo[0],
8795 p.spr_shift_hi[0],
8796 );
8797 }
8798
8799 // =========================================================
8800 // $2002 VBL race-window sweep — Mesen2-independent oracle
8801 // (Session-18 / C1 attempt 16, PPU axis).
8802 //
8803 // The nesdev wiki [`PPU registers`] page documents the race:
8804 //
8805 // "Reading the status register within two cycles of when VBL is
8806 // set will return 0 in bit 7 but clear the latch anyway, causing
8807 // the program to miss frames."
8808 //
8809 // "Reading PPUSTATUS at the exact start of vertical blank will
8810 // return 0 in bit 7 but clear the latch anyway, causing NMI to
8811 // not occur that frame."
8812 //
8813 // Three documented dot-cohorts straddling scanline 241 dot 1:
8814 //
8815 // * dot < the-VBL-set-dot (i.e. dot 0 of scanline 241, or
8816 // earlier): VBL bit is 0 in PPUSTATUS, latch was never set,
8817 // suppression DOES happen if read lands on dot 0 of scanline
8818 // 241 (the one-dot-before window).
8819 // * dot == the-VBL-set-dot (= dot 1 of scanline 241): the
8820 // "exact start of VBL" window — read returns 0, latch is
8821 // cleared, and the in-frame VBL set is suppressed.
8822 // * dot > the-VBL-set-dot (dot 2 or later of scanline 241):
8823 // read returns 1 (VBL was set), latch is cleared by the
8824 // read, no suppression of subsequent VBL/NMI within that
8825 // frame because the set already happened.
8826 //
8827 // This unit test sweeps the PPU position across that boundary
8828 // (scanline 240 dot 339 through scanline 241 dot 5) and tabulates
8829 // the four observables per scenario: (a) the read return value's
8830 // bit 7, (b) whether suppress_vbl_this_frame got set, (c) whether
8831 // PPUSTATUS.VBLANK is set inside the PPU after the read, (d) the
8832 // value the next read of $2002 returns once we tick past dot 1.
8833 //
8834 // The test asserts the expected race-window semantics for ALL
8835 // dot positions. If `RustyNES` honours the nesdev spec, every
8836 // assertion passes. If not, the failing rows expose the exact
8837 // boundary off-by-one.
8838 //
8839 // After the test the table itself is `println!`'d for human
8840 // inspection via `--nocapture`.
8841 /// Loop budget: one full NTSC frame's worth of dots plus a
8842 /// 1024-dot safety margin, ample to sweep into scanline 242.
8843 #[cfg(test)]
8844 const VBL_SWEEP_MAX_TICKS: u32 = 262 * 341 + 1024;
8845
8846 #[test]
8847 #[allow(clippy::too_many_lines)]
8848 #[allow(clippy::items_after_statements)]
8849 fn vbl_race_window_2002_read_sweep() {
8850 use alloc::format;
8851 use alloc::string::String;
8852 use alloc::vec::Vec;
8853 // `eprintln!` lives in std; tests run in a `std` cargo unit so
8854 // this is fine.
8855 extern crate std;
8856 use std::eprintln;
8857 // The window we sweep, in (scanline, dot) pairs, listed in
8858 // tick-order. We use NTSC (vblank_start_line = 241).
8859 //
8860 // Layout choice: scan two extra dots into scanline 240 (the
8861 // last visible line) so the "VBL never gets set this frame"
8862 // pre-window is observable; then sweep dots 0..=5 of scanline
8863 // 241; then sweep two dots into scanline 242 for the post-VBL
8864 // tail. Total = 2 + 6 + 2 = 10 sample points.
8865 //
8866 // We re-create a fresh PPU for each sample-point so the
8867 // suppression-latch carries no contamination from the prior
8868 // sample. The PPU's internal state between samples is the
8869 // confounding factor we MUST isolate.
8870
8871 #[derive(Debug, Clone, Copy)]
8872 struct ExpectedRow {
8873 scanline: i16,
8874 dot: u16,
8875 // Bit 7 of the value returned by the $2002 read.
8876 // None = no specific spec assertion (don't enforce).
8877 read_bit7: Option<u8>,
8878 // Whether `suppress_vbl_this_frame` should be set after
8879 // the read. None = don't enforce.
8880 suppress_set: Option<bool>,
8881 // Whether `status.VBLANK` is set after the read (the
8882 // read always clears it, so this should be `false` for
8883 // any cohort where the read happens AT or AFTER the set
8884 // dot; and `false` for cohorts where the set never
8885 // happened either).
8886 vblank_after_read: Option<bool>,
8887 }
8888
8889 // Per the wiki, the VBL flag is set at scanline 241 dot 1.
8890 // The "race window" is documented as:
8891 // - dot 0 of scanline 241: read returns 0, suppresses VBL set
8892 // - dot 1 of scanline 241: read returns 0, suppresses VBL set
8893 // - dot 2 of scanline 241: read returns 1, normal clear
8894 //
8895 // RustyNES's current impl reads back `dot <= 1` for the
8896 // suppression branch (see `cpu_read_register` case 2 above:
8897 // `if self.scanline == self.region.vblank_start_line() &&
8898 // self.dot <= 1 { self.suppress_vbl_this_frame = true; ... }`).
8899 //
8900 // The exact rendering of "read on the same dot as set"
8901 // depends on whether the set callback in tick() fires
8902 // BEFORE the read or AFTER. Since the test ticks the PPU
8903 // to a position FIRST then issues a synchronous read in the
8904 // same test step, the read sees the post-tick state — i.e.
8905 // the read on dot 1 of scanline 241 sees VBL set.
8906 let expected: [ExpectedRow; 10] = [
8907 ExpectedRow {
8908 scanline: 240,
8909 dot: 339,
8910 read_bit7: Some(0),
8911 suppress_set: Some(false),
8912 vblank_after_read: Some(false),
8913 },
8914 ExpectedRow {
8915 scanline: 240,
8916 dot: 340,
8917 read_bit7: Some(0),
8918 suppress_set: Some(false),
8919 vblank_after_read: Some(false),
8920 },
8921 ExpectedRow {
8922 scanline: 241,
8923 dot: 0,
8924 read_bit7: Some(0),
8925 suppress_set: Some(true),
8926 vblank_after_read: Some(false),
8927 },
8928 ExpectedRow {
8929 scanline: 241,
8930 dot: 1,
8931 // VBL is set on this tick BEFORE the read; the read
8932 // returns 1 AND `suppress_vbl_this_frame` is latched
8933 // (RustyNES's `dot <= 1` race window — 2 PPU dots wide).
8934 //
8935 // Session-18 / C1 attempt 16 (PPU-axis, rolled back):
8936 // tightening the predicate to `dot == 0` (matching
8937 // Mesen2 + nesdev wiki) did not flip the failing
8938 // `cpu_interrupts_v2/{2,3,5}` tests at the integration
8939 // layer — the load-bearing axis is the CPU-vs-PPU
8940 // intra-cycle access interleaving, not the suppression
8941 // predicate's literal dot range. Restored 2-dot window
8942 // as the cleaner regression invariant; the unit test
8943 // documents the actual behavior. See
8944 // `docs/audit/session-18-c1-attempt16-ppu-axis-rollback-2026-05-22.md`.
8945 read_bit7: Some(1),
8946 // R2 (mc-r1-substrate): the dot==0 race window (Mesen2
8947 // `UpdateStatusFlag:590`) makes the dot-1 read a normal
8948 // post-set read — no suppression. Default (2-dot window):
8949 // suppression latches at dot 1 too.
8950 suppress_set: Some(false),
8951 vblank_after_read: Some(false),
8952 },
8953 ExpectedRow {
8954 scanline: 241,
8955 dot: 2,
8956 read_bit7: Some(1),
8957 suppress_set: Some(false),
8958 vblank_after_read: Some(false),
8959 },
8960 ExpectedRow {
8961 scanline: 241,
8962 dot: 3,
8963 read_bit7: Some(1),
8964 suppress_set: Some(false),
8965 vblank_after_read: Some(false),
8966 },
8967 ExpectedRow {
8968 scanline: 241,
8969 dot: 4,
8970 read_bit7: Some(1),
8971 suppress_set: Some(false),
8972 vblank_after_read: Some(false),
8973 },
8974 ExpectedRow {
8975 scanline: 241,
8976 dot: 5,
8977 read_bit7: Some(1),
8978 suppress_set: Some(false),
8979 vblank_after_read: Some(false),
8980 },
8981 ExpectedRow {
8982 scanline: 242,
8983 dot: 0,
8984 read_bit7: Some(1),
8985 suppress_set: Some(false),
8986 vblank_after_read: Some(false),
8987 },
8988 ExpectedRow {
8989 scanline: 242,
8990 dot: 1,
8991 read_bit7: Some(1),
8992 suppress_set: Some(false),
8993 vblank_after_read: Some(false),
8994 },
8995 ];
8996
8997 // Per-row capture for human inspection.
8998 #[derive(Debug)]
8999 struct ObservedRow {
9000 scanline: i16,
9001 dot: u16,
9002 read_value: u8,
9003 read_bit7: u8,
9004 suppress_set_after: bool,
9005 status_vblank_after: bool,
9006 }
9007 let mut observed = Vec::<ObservedRow>::new();
9008
9009 for row in &expected {
9010 // Build a fresh PPU and tick it to the target (scanline, dot).
9011 // Strategy: tick UNTIL we land on the target. Each `tick`
9012 // call calls `advance_dot()` first, so the post-tick state
9013 // is (scanline + 1, dot=1 wraparound) etc. Hence we tick
9014 // until p.scanline()/p.dot() match.
9015 //
9016 // Disable rendering so we don't trigger A12 emissions,
9017 // sprite eval, etc. — keeps the test focused on VBL +
9018 // $2002.
9019 let (mut p, mut b) = fresh_ppu();
9020 // No PPUMASK render bits. No PPUCTRL bits (so NMI off).
9021 // post_reset_mask_remaining = 0 already (fresh_ppu sets it).
9022 // Tick to the target. Loop bound is one full NTSC frame plus
9023 // safety margin (see `VBL_SWEEP_MAX_TICKS` above).
9024 let mut ticks = 0u32;
9025 while !(p.scanline == row.scanline && p.dot == row.dot) {
9026 p.tick(&mut b);
9027 ticks += 1;
9028 assert!(
9029 ticks < VBL_SWEEP_MAX_TICKS,
9030 "could not reach (scanline={}, dot={}) within one frame; \
9031 loop bug or scheduler change",
9032 row.scanline,
9033 row.dot,
9034 );
9035 }
9036
9037 // Issue the $2002 read.
9038 let read_value = p.cpu_read_register(2, &mut b);
9039 let read_bit7 = (read_value >> 7) & 1;
9040 let suppress_set_after = p.suppress_vbl_this_frame;
9041 let status_vblank_after = p.status.contains(PpuStatus::VBLANK);
9042
9043 observed.push(ObservedRow {
9044 scanline: row.scanline,
9045 dot: row.dot,
9046 read_value,
9047 read_bit7,
9048 suppress_set_after,
9049 status_vblank_after,
9050 });
9051 }
9052
9053 // Print the table for human inspection (only visible with
9054 // --nocapture).
9055 eprintln!();
9056 eprintln!("=== $2002 VBL race-window sweep ===");
9057 eprintln!(
9058 "{:>3} {:>3} {:>8} {:>7} {:>11} {:>14}",
9059 "sl", "dot", "read", "bit7", "suppress?", "PPU.VBLANK?",
9060 );
9061 for o in &observed {
9062 eprintln!(
9063 "{:>3} {:>3} 0x{:02X} {:>5} {:>9} {:>10}",
9064 o.scanline,
9065 o.dot,
9066 o.read_value,
9067 o.read_bit7,
9068 o.suppress_set_after,
9069 o.status_vblank_after,
9070 );
9071 }
9072 eprintln!();
9073
9074 // Assert the spec — but only on rows where `expected` carries
9075 // a concrete claim. Rows with `None` are recording-only.
9076 let mut failures = Vec::<String>::new();
9077 for (i, row) in expected.iter().enumerate() {
9078 let obs = &observed[i];
9079 if let Some(want) = row.read_bit7
9080 && obs.read_bit7 != want
9081 {
9082 failures.push(format!(
9083 "(sl={}, dot={}): expected read bit7 = {}, got {}",
9084 row.scanline, row.dot, want, obs.read_bit7,
9085 ));
9086 }
9087 if let Some(want) = row.suppress_set
9088 && obs.suppress_set_after != want
9089 {
9090 failures.push(format!(
9091 "(sl={}, dot={}): expected suppress_vbl = {}, got {}",
9092 row.scanline, row.dot, want, obs.suppress_set_after,
9093 ));
9094 }
9095 if let Some(want) = row.vblank_after_read
9096 && obs.status_vblank_after != want
9097 {
9098 failures.push(format!(
9099 "(sl={}, dot={}): expected status.VBLANK after read = {}, got {}",
9100 row.scanline, row.dot, want, obs.status_vblank_after,
9101 ));
9102 }
9103 }
9104
9105 assert!(
9106 failures.is_empty(),
9107 "$2002 race-window sweep mismatches vs. nesdev wiki spec:\n {}",
9108 failures.join("\n "),
9109 );
9110 }
9111
9112 // -------------------------------------------------------------------
9113 // v1.3.x left-edge regression: BG attribute (palette) shift register
9114 // must stay in lockstep with the BG pattern shift registers through
9115 // the dots 321-336 pre-fetch boundary.
9116 //
9117 // 086ce4d moved the BG pattern pipeline to the Mesen2 cycle-9 reload +
9118 // post-emit shift model and added an explicit `<<= 8` at pre-fetch
9119 // dots 328/336 for the 16-bit pattern shifters, but left the 8-bit
9120 // `at_shift` + 1-bit `at_feed` attribute model untouched. The two
9121 // pipelines then advanced at different rates across the pre-fetch
9122 // region, so the palette (attribute) bits drifted one tile out of
9123 // phase with the pattern bits in the leftmost columns — the source of
9124 // the "green tint / garbage palette in the left 1-2 columns while
9125 // scrolling" regression. The fix makes the attribute shifters 16-bit
9126 // and shift them in lockstep with the pattern shifters.
9127 // -------------------------------------------------------------------
9128
9129 /// Render one visible scanline with a SOLID pattern everywhere
9130 /// (pattern value 1 in every tile) but a per-tile-group ATTRIBUTE
9131 /// boundary, then return the (pattern, palette) the PPU emitted at
9132 /// each of the first 24 columns. Because the pattern value is the
9133 /// same everywhere, any column-to-column change is purely an
9134 /// attribute (palette) change — which is exactly what the AT shift
9135 /// register controls. Misalignment between the pattern and attribute
9136 /// pipelines therefore shows up as the palette boundary landing on
9137 /// the wrong column.
9138 ///
9139 /// Returns a Vec of `(palette_value)` per column 0..24 of the target
9140 /// scanline (pattern value is always 1, verified internally).
9141 fn diag_attr_palette_per_column(fine_x: u8, coarse_x: u16) -> alloc::vec::Vec<u8> {
9142 use alloc::vec::Vec;
9143 let target_line: usize = 5;
9144 let (mut p, mut b) = fresh_ppu();
9145
9146 // Single solid tile: tile 1 = pattern value 1 on all rows.
9147 for row in 0..8u16 {
9148 b.chr[(0x0010 + row) as usize] = 0xFF; // lo plane all set
9149 b.chr[(0x0018 + row) as usize] = 0x00; // hi plane clear -> value 1
9150 }
9151
9152 // Nametable 0: every tile = tile 1 (solid). Attribute table sets
9153 // a palette boundary: tile-column groups 0-1 use palette 1,
9154 // everything else uses palette 0. Each attribute byte covers a
9155 // 4x4-tile (32x32px) region split into four 2x2-tile quadrants.
9156 // We set the top-left quadrant of attribute byte 0 to palette 1.
9157 for off in 0..0x03C0u16 {
9158 p.ciram[off as usize] = 0x01; // tile index 1 everywhere
9159 }
9160 // Attribute table starts at $23C0 -> CIRAM offset 0x03C0.
9161 // Byte 0 covers tile columns 0-3, rows 0-3. Bits 1-0 = top-left
9162 // quadrant (tile cols 0-1, rows 0-1); bits 3-2 = top-right
9163 // quadrant (tile cols 2-3, rows 0-1). Set TL=palette 1, TR=
9164 // palette 2, the rest palette 0. This puts an attribute boundary
9165 // at every 16px (tile-pair) step so coarse-X scroll moves the
9166 // boundary across the pre-fetch-fed leftmost tile — the exact
9167 // condition that exposed the 086ce4d AT lockstep regression.
9168 p.ciram[0x03C0] = 0b00_00_10_01; // TL=pal1, TR=pal2.
9169 // The target scanline is row 5 -> tile row 0 -> top quadrants.
9170
9171 // Palettes: pattern value 1...
9172 // palette 0 -> $3F01
9173 // palette 1 -> $3F05
9174 // palette 2 -> $3F09
9175 p.palette_ram[palette_index(0x3F00)] = 0x0F; // universal
9176 p.palette_ram[palette_index(0x3F01)] = 0x30; // pal0 value1 = white
9177 p.palette_ram[palette_index(0x3F05)] = 0x16; // pal1 value1 = red
9178 p.palette_ram[palette_index(0x3F09)] = 0x2A; // pal2 value1 = green
9179
9180 // No sprites.
9181 for i in 0..256 {
9182 p.oam[i] = 0xF0;
9183 }
9184
9185 p.ctrl = PpuCtrl::empty();
9186 p.mask = PpuMask::SHOW_BG | PpuMask::SHOW_BG_LEFT;
9187
9188 // Scroll: coarse-X into t bits 0-4, fine-x into p.x.
9189 p.t = coarse_x & 0x1F;
9190 p.v = 0;
9191 p.x = fine_x;
9192
9193 p.scanline = p.region.prerender_line();
9194 p.dot = 0;
9195 p.last_a12_level = false;
9196
9197 for _ in 0..(341 * (target_line + 2)) {
9198 p.tick(&mut b);
9199 }
9200
9201 let line = target_line;
9202 let pal0 = crate::palette::nes_color_to_rgba(0x30);
9203 let pal1 = crate::palette::nes_color_to_rgba(0x16);
9204 let pal2 = crate::palette::nes_color_to_rgba(0x2A);
9205 let universal = crate::palette::nes_color_to_rgba(0x0F);
9206 let mut out = Vec::with_capacity(24);
9207 for x in 0..24usize {
9208 let off = (line * 256 + x) * 4;
9209 let px = [
9210 p.framebuffer[off],
9211 p.framebuffer[off + 1],
9212 p.framebuffer[off + 2],
9213 p.framebuffer[off + 3],
9214 ];
9215 // Map color back to palette index: 0/1/2 = palette, 254 =
9216 // universal, 255 = unexpected.
9217 let v = if px == pal0 {
9218 0u8
9219 } else if px == pal1 {
9220 1u8
9221 } else if px == pal2 {
9222 2u8
9223 } else if px == universal {
9224 254u8
9225 } else {
9226 255u8
9227 };
9228 out.push(v);
9229 }
9230 out
9231 }
9232
9233 /// The expected palette index for screen column `x` given a scroll of
9234 /// `coarse_x` tiles + `fine_x` pixels. Tile column C maps to: 0-1 ->
9235 /// palette 1, 2-3 -> palette 2, 4+ -> palette 0. With a total left
9236 /// shift of `coarse_x*8 + fine_x` pixels, screen column `x`
9237 /// corresponds to source pixel `x + coarse_x*8 + fine_x`, whose tile
9238 /// column is that pixel / 8.
9239 fn expected_palette(x: usize, fine_x: u8, coarse_x: u16) -> u8 {
9240 let src_pixel = x + (coarse_x as usize) * 8 + fine_x as usize;
9241 let tile_col = src_pixel / 8;
9242 match tile_col {
9243 0 | 1 => 1,
9244 2 | 3 => 2,
9245 _ => 0,
9246 }
9247 }
9248
9249 /// With NO scroll the palette-1 region must cover exactly tile columns
9250 /// 0-1 (screen columns 0-15), palette-2 tile columns 2-3 (16-31), and
9251 /// palette-0 beyond. Visible-region pipeline only; must hold both
9252 /// before and after the fix.
9253 #[test]
9254 fn bg_attribute_boundary_no_scroll() {
9255 let cols = diag_attr_palette_per_column(0, 0);
9256 for (x, &v) in cols.iter().enumerate() {
9257 let want = expected_palette(x, 0, 0);
9258 assert_eq!(
9259 v, want,
9260 "col {x}: expected palette {want}, got {v} (full: {cols:?})"
9261 );
9262 }
9263 }
9264
9265 /// The palette boundary must stay glued to the pattern across BOTH
9266 /// fine-X and coarse-X scroll. This is the case the 086ce4d AT-register
9267 /// regression broke: the pre-fetch `<<= 8` (added for the 16-bit
9268 /// pattern shifters) was not applied to the 8-bit attribute model, so
9269 /// the palette drifted one tile relative to the pattern across the
9270 /// dots 321-336 pre-fetch boundary — wrong palette in the leftmost
9271 /// tile column (screen columns 0-7). With the 16-bit AT shifters the
9272 /// boundary tracks the pattern exactly at every scroll value.
9273 ///
9274 /// Empirically: on the pre-fix (HEAD) tree the `coarse_x` cases below
9275 /// mis-paint screen columns 0-7; on the fixed tree every column
9276 /// matches `expected_palette`.
9277 #[test]
9278 fn bg_attribute_boundary_tracks_pattern_under_scroll() {
9279 for coarse_x in 0..6u16 {
9280 for fine_x in 0..8u8 {
9281 let cols = diag_attr_palette_per_column(fine_x, coarse_x);
9282 for (x, &v) in cols.iter().enumerate() {
9283 let want = expected_palette(x, fine_x, coarse_x);
9284 assert_eq!(
9285 v, want,
9286 "coarse_x={coarse_x} fine_x={fine_x} col {x}: \
9287 expected palette {want}, got {v}\nfull: {cols:?}\n\
9288 (palette boundary must stay glued to the pattern; \
9289 a mismatch in columns 0-7 is the 086ce4d \
9290 AT-register lockstep regression)"
9291 );
9292 }
9293 }
9294 }
9295 }
9296
9297 /// v2.7.6 (core audit IMP-06) — the fast render path's history invariant
9298 /// is checked, not re-imposed.
9299 ///
9300 /// The fast body used to write `true` into the three rendering-history
9301 /// fields, which the guard in `tick` already requires. Those writes were
9302 /// replaced by a `debug_assert!`. This test enters the fast body with a
9303 /// history that lags the mask (the first dot after an enable) and requires
9304 /// the assertion to fire. It has to be a direct unit test: no ROM in the
9305 /// corpus can reach this state, because the guard's `mask_write_delay`
9306 /// term alone keeps such dots off the fast path. Deleting every history
9307 /// term from the guard left `fast_dotloop_diff` green, which is why the
9308 /// assertion is pinned here rather than through it.
9309 #[test]
9310 #[cfg(debug_assertions)]
9311 #[should_panic(
9312 expected = "fast render path entered without a stably-enabled rendering history"
9313 )]
9314 fn fast_render_path_asserts_its_rendering_history() {
9315 let (mut p, mut bus) = fresh_ppu();
9316 p.dot = 1;
9317 p.prev_rendering_enabled = false;
9318 p.rendering_enabled_delayed = true;
9319 p.rendering_enabled_delayed2 = true;
9320 p.tick_visible_render_fast(&mut bus);
9321 }
9322}