From 4a00075a523682b71ec1e785dd02c8f83ba48ca1 Mon Sep 17 00:00:00 2001 From: Raphael Amorim Date: Fri, 5 Jun 2026 09:20:14 +0200 Subject: [PATCH] drop reference comments --- bus.jam | 16 ++-- cdrom.jam | 130 +++++++++++++-------------- cpu.jam | 82 +++++++++-------- cue.jam | 20 ++--- disc.jam | 4 +- dma.jam | 78 ++++++++-------- gpu.jam | 141 ++++++++++++++--------------- gte.jam | 259 ++++++++++++++++++++++++++---------------------------- irq.jam | 4 +- main.jam | 85 +++++++++--------- mcd.jam | 10 +-- mdec.jam | 47 +++++----- pad.jam | 20 ++--- sdl.jam | 4 +- sio1.jam | 3 +- spu.jam | 74 ++++++++-------- tests.jam | 24 ++--- timer.jam | 54 ++++++------ xa.jam | 18 ++-- 19 files changed, 523 insertions(+), 550 deletions(-) diff --git a/bus.jam b/bus.jam index 3457eb4..a2d72c9 100644 --- a/bus.jam +++ b/bus.jam @@ -1,4 +1,4 @@ -// PSX memory bus (port of psxe/psx/bus.c). +// PSX memory bus. // // Routes word / halfword / byte reads and writes through the masked // physical address. Owns the RAM / BIOS / scratchpad / flat-IO / VRAM @@ -117,14 +117,14 @@ pub fn bufWrite8(buf: *mut[] u8, off: u32, val: u32) { buf[off] = (val & 0xFF) as u8; } -// Per-region routing. Same as psxe's static IO-base/size table inlined +// Per-region routing. A static IO-base/size table inlined // into an address comparator. `kind` values: // 0 = unmapped (reads return 0) // 1 = RAM // 2 = BIOS ROM // 3 = scratchpad // 4 = I/O register file -// 5 = EXP1 (returns 0xFF on read — psxe psx_exp1_read* with no cart loaded) +// 5 = EXP1 (returns 0xFF on read with no cart loaded) // 6 = EXP2 (DTL debug-console region; only the ATC status byte matters) pub fn route(paddr: u32) BusRoute { if (paddr < 0x00800000) { @@ -143,8 +143,8 @@ pub fn route(paddr: u32) BusRoute { const r = BusRoute { kind: 4, off: paddr - 0x1F801000 }; return r; } - // EXP2 (0x1F802000–0x1F9FFFFF): psxe routes this whole range to the - // EXP2/DTL device (exp2.h). Previously it fell into the I/O branch and + // EXP2 (0x1F802000–0x1F9FFFFF): this whole range routes to the + // EXP2/DTL device. Previously it fell into the I/O branch and // read back the zero buffer; route it as kind 6 so the DTL ATC status // byte reads correctly. if (paddr >= 0x1F802000 && paddr < 0x1FA00000) { @@ -175,7 +175,7 @@ pub fn allocBus() Bus { var io = Vec(u8).filled(0, IO_SIZE); var vram = Vec(u8).filled(0, VRAM_SIZE); // Pre-init MC1 (0x1F801000..0x1F801023) and MC2 (0x1F801060) with the - // BIOS-expected default values per psxe mc1.c/mc2.c. The BIOS reads + // BIOS-expected default values. The BIOS reads // these at boot to verify access timings and RAM mirror config; before // it writes them itself the values must match real hardware so the // self-check doesn't latch a "?" error. @@ -332,7 +332,7 @@ pub fn busRead32(bus: Bus, addr: u32) u32 { 2 { return bufRead32(bus.bios.ptr, r.off & (BIOS_SIZE - 1)); } 3 { return bufRead32(bus.spad.ptr, r.off & (SPAD_SIZE - 1)); } 4 { return ioRead32(bus, r.off); } - // kind 5: EXP1 (no cart loaded) — psxe exp1.c reads return 0xFF per byte. + // kind 5: EXP1 (no cart loaded) — reads return 0xFF per byte. 5 { return 0xFFFFFFFF; } _ { return 0; } } @@ -362,7 +362,7 @@ pub fn busRead8(bus: Bus, addr: u32) u32 { 5 { return 0xFF; } // EXP2 DTL: ATC_STAT (off 0) reads atc_stat|8; jam never drives // the DTL TTY so atc_stat is 0, making it a constant 8. ATC_DATA - // (off 2) and the rest read 0 (psxe exp2.c psx_exp2_read8:38-48). + // (off 2) and the rest read 0. 6 { if (r.off == 0) { return 8; } return 0; } _ { return 0; } } diff --git a/cdrom.jam b/cdrom.jam index eb1ba63..b8a7ab7 100644 --- a/cdrom.jam +++ b/cdrom.jam @@ -1,4 +1,4 @@ -// CD-ROM controller — minimal port of psxe `psx/dev/cdrom/*.c` covering +// CD-ROM controller — a minimal implementation covering // the BIOS-boot subset of commands plus the FIFO + IRQ + state-machine // plumbing every command depends on. Audio (CDDA) is stubbed; XA-ADPCM // streams through to the SPU CD-audio FIFO when MODE.XA is set. @@ -6,7 +6,7 @@ // Register map at 0x1F801800..0x1F801803 — every register is bank- // selected: writes to address 0x800 set bits 0-1 of the bank index, // then addresses 0x801..0x803 take different meanings per bank. The -// dispatch table mirrors psxe (psx_cdrom_write8 / cdrom_read8): +// dispatch table: // // bank read 0x800 status (busy/parm-empty/resp-empty/...) // bank read 0x801 pop one byte from the response FIFO @@ -38,15 +38,15 @@ const { spuPushCdSample } = import("spu"); const SECTOR_BYTES: u32 = 2352; -// State machine states (psxe enum: IDLE/TX_RESP1/TX_RESP2/READ/PLAY). +// State machine states (IDLE/TX_RESP1/TX_RESP2/READ/PLAY). const CD_STATE_IDLE: u32 = 0; const CD_STATE_TX_RESP1: u32 = 1; const CD_STATE_TX_RESP2: u32 = 2; const CD_STATE_READ: u32 = 3; const CD_STATE_PLAY: u32 = 4; -// psxe's CD_DELAY_FR / CD_DELAY_INIT_FR / CD_DELAY_READ_* (CPU cycles). -const CD_DELAY_1MS: u32 = 33869; // psxe CD_DELAY_1MS +// Response / read scheduling delays (CPU cycles). +const CD_DELAY_1MS: u32 = 33869; // 1ms at 33.8688MHz const CD_DELAY_FR: u32 = 50401; const CD_DELAY_INIT_FR: u32 = 81102; const CD_DELAY_READ_SS: u32 = 451584; // 33_868_800 / 75 @@ -140,7 +140,7 @@ pub const Cdrom = struct { delay: u32, lba: u32, pendingLba: u32, - pendingSpeedSwitch: u32, // psxe pending_speed_switch_delay (1x<->2x resync) + pendingSpeedSwitch: u32, // 1x<->2x speed-change resync delay dataRidx: u32, dataWidx: u32, // XA-ADPCM per-channel history (2 i32 each). Persists across @@ -148,7 +148,7 @@ pub const Cdrom = struct { xaLh: [2]i32, xaRh: [2]i32, // Last resampled OUTPUT sample per channel — the `ls` seed for the next - // sector's XA→44.1kHz rate conversion (psxe xa_prev_left/right_sample). + // sector's XA→44.1kHz rate conversion. xaPrevL: i32, xaPrevR: i32, // Parameter / response FIFOs. Each is a 32-byte ring with u8 @@ -171,8 +171,8 @@ pub const Cdrom = struct { seekPrec: 1, prevSpeed: 0, paramR: 0, paramW: 0, respR: 0, respW: 0, - // Start at LBA 150 (00:02:00, first data sector) like psxe - // (psx_cdrom_init). jam's LBA is pregap-inclusive — cueRead + // Start at LBA 150 (00:02:00, first data sector). + // jam's LBA is pregap-inclusive — cueRead // maps off=(lba-startLba)*SECTOR with track1 startLba=150 — // so 0 would read a negative offset (zero pregap), not sector 0. delay: 0, lba: 150, pendingLba: 150, pendingSpeedSwitch: 0, @@ -248,7 +248,7 @@ pub const Cdrom = struct { // status byte - // Mirrors psxe cdrom_get_stat (cdrom.c:454): SPINDLE always, plus + // Status byte: SPINDLE always, plus // READ while the sector pump is active, SHELLOPEN if no disc, // IDERROR if mode bit 4 is set. pub fn getStat(self: mut Self) u32 { @@ -264,8 +264,8 @@ pub const Cdrom = struct { // Status register at offset 0 (always, regardless of bank). pub fn readStatus(self: mut Self) u32 { var r: u32 = (self.index as u32) & 0x3; - // bit 2 (ADPBUSY): XA-ADPCM streaming active — psxe cdrom_read_status - // returns `xa_playing << 2`. Set when a ReadN/ReadS runs in XA mode. + // bit 2 (ADPBUSY): XA-ADPCM streaming active. Set when a ReadN/ReadS + // runs in XA mode. // (Brave Fencer's FMV polls this; jam hardcoding 0 made the game // branch away from the decode path — the SECOND trace-diff divergence, // instr ~368.37M.) @@ -323,14 +323,14 @@ pub const Cdrom = struct { var delay: u32 = CD_DELAY_FR; if ((val & 0xFF) == CDL_INIT) { delay = CD_DELAY_INIT_FR; } self.delay = delay; - // psxe's cdrom_write_cmd checks `if (state == CD_STATE_READ) busy=1` - // (cdrom.c:683) but `state` was just set to TX_RESP1 two lines above - // (cdrom.c:669), so that branch is ALWAYS FALSE — psxe never raises - // BUSYSTS from a command write. We must not either: asserting busy on - // the mid-read GetlocL/Getstat polls that the BFM/Suikoden FMV loader - // spams makes the game see the drive busy where psxe shows it idle, so - // it re-reads the resource sectors forever and never triggers the MDEC - // decode (FMV black-screen). + // A command write does NOT raise BUSYSTS: `state` was just set + // to TX_RESP1 two lines above, so any "if reading, mark busy" + // check sees TX_RESP1, not READ, and never fires. We must not + // assert busy either: doing so on the mid-read GetlocL/Getstat + // polls that the BFM/Suikoden FMV loader spams makes the game + // see the drive busy when it should be idle, so it re-reads the + // resource sectors forever and never triggers the MDEC decode + // (FMV black-screen). } // Push parameter byte. 2 { self.pushParam(val); } @@ -361,17 +361,17 @@ pub const Cdrom = struct { // state-machine tick - // Handle the command response — equivalent to psxe's - // `cdrom_cmd_table[cmd](cdrom)` running after the TX_RESP1 delay. + // Handle the command response — dispatch the queued command + // after the TX_RESP1 delay elapses. pub fn executeCommand(self: mut Self, disc: *mut[] Disc, ic: *mut[] u32, cop0: *mut[] u32) { const cmd: u32 = self.pendingCmd as u32; self.busy = 0; - // psxe cdrom_handle_resp1 prechecks (cdrom.c:271-447), before - // dispatch: (1) disc-required commands with no disc → INT5(11h,80h), + // Response-1 prechecks, before dispatch: + // (1) disc-required commands with no disc → INT5(11h,80h), // (3) wrong parameter count → INT5(03h,20h), unknown command → - // INT5(03h,40h). (Stage 2 version checks are omitted — jam models a + // INT5(03h,40h). (Version checks are omitted — jam models a // single CD-ROM revision.) `pcnt` = parameters queued for this cmd. const pcnt: u32 = ((self.paramW as u32) - (self.paramR as u32)) & 0xFF; const c0: bool = cmd == CDL_GETSTAT || cmd == CDL_FORWARD || @@ -431,7 +431,7 @@ pub const Cdrom = struct { if (cmd == CDL_GETSTAT || cmd == CDL_MUTE || cmd == CDL_DEMUTE || cmd == CDL_SETMODE || cmd == CDL_RESET || cmd == CDL_SETLOC) { if (cmd == CDL_SETMODE) { - // psxe cmd_setmode (impl.c:355-361): a 1x<->2x speed change + // SetMode: a 1x<->2x speed change // costs a big ~650ms drive resync, charged to the next read. // FMV setup (SetMode double-speed -> ReadN) relies on it; // without it the first STR sector arrives too early. @@ -445,7 +445,7 @@ pub const Cdrom = struct { const m: u32 = self.popParam(); const s: u32 = self.popParam(); const f: u32 = self.popParam(); - // psxe validates VALID_MSF before accepting (impl.c:138-158): + // Validate the MSF before accepting: // BCD-valid nibbles, seconds < 0x60, frame < 0x75. Invalid // MSF → INT5(stat, 0x10) instead of an INT3 success. const bcdOk: bool = ((m & 0x0F) <= 9) && ((m >> 4) <= 9) && @@ -463,7 +463,7 @@ pub const Cdrom = struct { self.ifr = 3; self.pushResp(self.getStat()); if (cmd == CDL_SETMODE) { - // psxe impl.c:363 — SetMode hard-resets to IDLE. + // SetMode hard-resets the state machine to IDLE. self.state = CD_STATE_IDLE as u8; self.prevState = CD_STATE_IDLE as u8; self.readOngoing = 0; @@ -493,7 +493,7 @@ pub const Cdrom = struct { self.ifr = 3; self.pushResp(self.getStat()); self.state = CD_STATE_TX_RESP2 as u8; - // psxe cdrom_cmd_init (impl.c:308) sets the 2nd-response delay to + // Init's 2nd-response delay is // CD_DELAY_1MS (33869), NOT CD_DELAY_INIT_FR (81102). The 47233-cyc // gap made jam's Init INT2 fire late, slipping the CD IRQ at ~164.8M // (the next trace divergence after the GPU-acc/-ffast-math one). @@ -504,9 +504,9 @@ pub const Cdrom = struct { self.ifr = 3; self.pushResp(self.getStat()); self.state = CD_STATE_TX_RESP2 as u8; - // seek_precision is set in resp2 on success, like psxe — not - // here in resp1. psxe cdrom_cmd_seekl sets the 2nd-response delay - // to CD_DELAY_1MS (NOT FR) — the FR=50401 vs 1MS=33869 gap (16532 + // seek precision is set in resp2 on success, not + // here in resp1. SeekL's 2nd-response delay is + // CD_DELAY_1MS (NOT FR) — the FR=50401 vs 1MS=33869 gap (16532 // cyc) made jam's SeekL INT2 fire ~8266 instrs late, the root of // the FMV CD-INT divergence at ~102.7M. self.delay = CD_DELAY_1MS; @@ -530,7 +530,7 @@ pub const Cdrom = struct { self.state = CD_STATE_READ as u8; self.prevState = CD_STATE_READ as u8; self.readOngoing = 1; - // psxe impl.c:229/683 — a read in XA-ADPCM mode marks the drive + // A read in XA-ADPCM mode marks the drive // as XA-streaming (status bit2 ADPBUSY). The game polls this. if ((self.mode as u32 & MODE_XA_ADPCM) != 0) { self.xaPlaying = 1; } self.delay = self.readDelay(); @@ -566,7 +566,7 @@ pub const Cdrom = struct { } if (cmd == CDL_GETLOCP) { self.ifr = 3; - // psxe cdrom_cmd_getlocp (impl.c:418-459): track/index from + // GetLocP: track/index from // the disc; subtract the 25-sector seek slop FIRST, then derive // both the relative (within-track) and absolute MSF from it. const lbaRaw: u32 = self.lba; @@ -598,7 +598,7 @@ pub const Cdrom = struct { return; } if (cmd == CDL_GETTN) { - // psxe: stat, first-track = 1 (raw), last = ITOB(track count). + // GetTN: stat, first-track = 1 (raw), last = BCD(track count). self.ifr = 3; self.pushResp(self.getStat()); self.pushResp(1); @@ -612,7 +612,7 @@ pub const Cdrom = struct { self.errorOut(CD_STAT_SPINDLE, CD_ERR_INVALID_SUBFUNC); return; } - // psxe cdrom_cmd_gettd: look up the track's absolute LBA → + // GetTD: look up the track's absolute LBA → // MM:SS; INT5 if past the last track (TS_FAR = 0xFFFFFFFF here). const track: u32 = (bcd & 0x0F) + ((bcd >> 4) & 0x0F) * 10; const f: u32 = discTrackLba(disc, track); @@ -657,7 +657,7 @@ pub const Cdrom = struct { self.ifr = 3; self.pushResp(self.getStat()); self.state = CD_STATE_TX_RESP2 as u8; - self.delay = CD_DELAY_1MS; // psxe cdrom_cmd_seekp: 1MS, not FR + self.delay = CD_DELAY_1MS; // SeekP: 1MS, not FR return; } @@ -722,10 +722,10 @@ pub const Cdrom = struct { return; } if (cmd == CDL_SEEKL) { - // psxe (impl.c:541-569): query the seek TARGET (pending_lba) + // Query the seek TARGET (pendingLba) // FIRST; TS_FAR → INVALID_SUBFUNC, TS_AUDIO → SEEK_FAILED, and // the seek is NOT committed on error. On success commit and - // set seek_precision = 1. + // set seek precision = 1. const tsl: u32 = discQuery(disc, self.pendingLba); if (tsl == 3) { self.errorOut(CD_STAT_SPINDLE | CD_STAT_SEEKERROR, @@ -749,8 +749,8 @@ pub const Cdrom = struct { return; } if (cmd == CDL_SEEKP) { - // psxe cdrom_cmd_seekp (impl.c:577-587): INT2(stat), commit - // the seek, seek_precision = 0. No seek-error check (SeekP is + // SeekP: INT2(stat), commit + // the seek, seek precision = 0. No seek-error check (SeekP is // the audio-seek command and may land on CDDA tracks). self.ifr = 2; self.pushResp(self.getStat()); @@ -764,7 +764,7 @@ pub const Cdrom = struct { return; } if (cmd == CDL_STOP) { - // psxe cdrom_cmd_stop (impl.c:267-278): rewind to LBA 150 and + // Stop: rewind to LBA 150 and // pause, then INT2(stat). self.pendingLba = 150; self.processSetloc(); @@ -777,7 +777,7 @@ pub const Cdrom = struct { return; } if (cmd == CDL_MOTORON || cmd == CDL_SETSESSION || cmd == CDL_READTOC) { - // psxe: each sends a second INT2(stat) (impl.c:250,471,725). + // Each sends a second INT2(stat). self.ifr = 2; self.pushResp(self.getStat()); self.state = CD_STATE_IDLE as u8; @@ -841,7 +841,7 @@ pub const Cdrom = struct { self.busy = 0; } - // Per-scanline tick. Mirror of psxe's psx_cdrom_update. + // Per-scanline tick. Advances the response/read state machine. pub fn update(self: mut Self, disc: *mut[] Disc, spu: *mut[] u8, ic: *mut[] u32, cop0: *mut[] u32, cyc: u32) { var delay: u32 = self.delay; @@ -870,15 +870,15 @@ pub const Cdrom = struct { } if (self.state == CD_STATE_TX_RESP1 as u8) { self.executeCommand(disc, ic, cop0); - // psxe cdrom.c:548-554: switching to READ after a non-read + // Switching to READ after a non-read // command in TX_RESP1 incurs a delay. if (self.state == CD_STATE_READ as u8) { self.processSetloc(); - // First sector after a command uses psxe's CD_DELAY_ONGOING_READ - // (= readDelay + 4ms). psxe COMPUTES a speed-switch resync delay - // but then OVERWRITES it with CD_DELAY_ONGOING_READ (cdrom.c:567), + // First sector after a command uses the ongoing-read delay + // (= readDelay + 4ms). The speed-switch resync delay is + // computed but then overwritten by this ongoing-read delay, // so it never applies; adding pendingSpeedSwitch (~650ms once) - // delayed the first FMV sector vs psxe. Drop it; just consume it. + // delayed the first FMV sector. Drop it; just consume it. self.delay = self.readDelay() + 4 * 33869; self.pendingSpeedSwitch = 0; } @@ -886,11 +886,11 @@ pub const Cdrom = struct { self.executeResp2(disc); if (self.state == CD_STATE_READ as u8) { self.processSetloc(); - // First sector after a command uses psxe's CD_DELAY_ONGOING_READ - // (= readDelay + 4ms). psxe COMPUTES a speed-switch resync delay - // but then OVERWRITES it with CD_DELAY_ONGOING_READ (cdrom.c:567), + // First sector after a command uses the ongoing-read delay + // (= readDelay + 4ms). The speed-switch resync delay is + // computed but then overwritten by this ongoing-read delay, // so it never applies; adding pendingSpeedSwitch (~650ms once) - // delayed the first FMV sector vs psxe. Drop it; just consume it. + // delayed the first FMV sector. Drop it; just consume it. self.delay = self.readDelay() + 4 * 33869; self.pendingSpeedSwitch = 0; } @@ -905,8 +905,8 @@ pub const Cdrom = struct { pub fn handleRead(self: mut Self, disc: *mut[] Disc, spu: *mut[] u8) { self.processSetloc(); if (self.hasDisc == 0) { - // psxe reports no-disc as SHELLOPEN (0x10), giving INT5(11h,80h) - // (cdrom_handle_resp1:300) — not SPINDLE (which gave 03h,80h). + // Report no-disc as SHELLOPEN (0x10), giving INT5(11h,80h) + // — not SPINDLE (which would give 03h,80h). self.errorOut(CD_STAT_SHELLOPEN, CD_ERR_NO_DISC); return; } @@ -917,9 +917,9 @@ pub const Cdrom = struct { return; } // XA-ADPCM audio sectors are decoded to the SPU, but we STILL deliver - // the sector via INT1 + the data FIFO. psxe's cdrom_handle_read raises - // INT1 for EVERY data-track sector regardless of submode (it decodes - // XA on a separate parallel xa_lba path), and so does real hardware + // the sector via INT1 + the data FIFO. INT1 fires for EVERY data-track + // sector regardless of submode (XA is decoded on a separate parallel + // path), as on real hardware // when no audio filter is set. Suppressing audio-sector INT1s here // (the previous "filter" behaviour, ported from Avocado's filtered // model) desynced games that do SOFTWARE de-interleaving — they expect @@ -944,7 +944,7 @@ pub const Cdrom = struct { self.ifr = 1; self.pushResp(self.getStat()); - self.pendingLba = lba + 1; // psxe cdrom_handle_read: deferred via pending_lba, lba promoted on next process_setloc (GetLocL/P returns current sector, not next) + self.pendingLba = lba + 1; // advance deferred via pendingLba; lba promoted on the next processSetloc (so GetLocL/P returns the current sector, not the next) self.delay = self.readDelay(); } @@ -980,13 +980,13 @@ pub const Cdrom = struct { // f18khz sector) to the SPU's 44100 Hz output rate before pushing. The // SPU CD FIFO is drained one entry per 44100 Hz output sample, so // pushing native-rate samples 1:1 plays XA ~17% too fast and overruns - // the FIFO (most samples dropped → warble). Faithful port of psxe - // cdrom_resample_xa_buf (audio.c:52-86): conceptually a 7×-upsample — + // the FIFO (most samples dropped → warble). The resampler is + // conceptually a 7×-upsample — // whose `(k+1)/8` integer-divides to 0, making it a sample-and-hold of // the previous source sample — then decimate by m=f18khz?3:6, yielding // `resampleCount` samples at 44100 Hz. Output j samples source index // q-1 where q=(j*m)/7; q==0 uses the previous sector's last output - // sample (psxe's `ls`, carried in xaPrevL/R). n (2016 stereo / 4032 + // sample (the `ls` seed, carried in xaPrevL/R). n (2016 stereo / 4032 // mono) bounds the source; q-1 never reaches it for any valid j. const f18: bool = xaSectorIs18kHz(scratch.asPtr()); var m: u32 = 6; @@ -1045,19 +1045,19 @@ pub const Cdrom = struct { // free helpers (no state required) -// Binary-to-BCD encoder (psxe ITOB). Inputs 0..99. +// Binary-to-BCD encoder. Inputs 0..99. pub fn bcdEnc(v: u32) u32 { const lo: u32 = v % 10; const hi: u32 = (v / 10) % 10; return (hi << 4) | lo; } -// BCD validity check (psxe VALID_BCD): both nibbles must be 0..9. +// BCD validity check: both nibbles must be 0..9. pub fn bcdValid(b: u32) bool { return ((b & 0x0F) <= 9) && (((b >> 4) & 0x0F) <= 9); } -// cdrom_version_id table (psxe), 4 bytes per version. Index 1 = C0A. +// CD-ROM controller version-ID table, 4 bytes per version. Index 1 = C0A. // `version` is unused for v1 — would index a table for other versions. pub fn versionByteForIdx(version: u32, n: u32) u32 { if (n == 0) { return 0x94; } diff --git a/cpu.jam b/cpu.jam index 7d0c3a7..190e67c 100644 --- a/cpu.jam +++ b/cpu.jam @@ -217,7 +217,7 @@ pub fn raiseException(c: mut Cpu, cop0: *mut[] u32, cause: u32) { pub fn cop0Rfe(cop0: *mut[] u32) { var sr: u32 = cop0Read(cop0, C0_SR); const mode: u32 = sr & 0x3F; - // Match psxe: clear only bits 0-3, preserve bits 4-5 (old-mode pair). + // Clear only bits 0-3, preserve bits 4-5 (old-mode pair). sr = (sr & 0xFFFFFFF0) | (mode >> 2); cop0Write(cop0, C0_SR, sr); } @@ -288,7 +288,7 @@ pub fn iAddi(c: mut Cpu, opc: u32, regs: *mut[] u32, cop0: *mut[] u32) { const r: u32 = s + imm; const ovf: u32 = (s ^ r) & (imm ^ r); if ((ovf & 0x80000000) != 0) { - // psxe cpu.c:465-481: skip the rd write on overflow so the + // Skip the rd write on overflow so the // exception handler sees the pre-state. raiseException(c, cop0, CAUSE_OV); return; @@ -369,16 +369,16 @@ pub fn iRegimm(c: mut Cpu, opc: u32, regs: *mut[] u32) { c.branch = 1; // Link only on the two explicit AL variants (0x10/0x11). For - // unknown rt values psxe's default path treats the op as a plain + // unknown rt values the op is treated as a plain // BLTZ/BGEZ depending on bit 16 — no link. if (variant == 0x10 || variant == 0x11) { gprWrite(regs, 31, c.nextPc); } - // psxe (cpu.c REGIMM default): branch type is decided by bit 16 - // (= rt & 1). Even-numbered rt → BLTZ semantics, odd → BGEZ. Our - // previous code only matched 0x00/0x10 for BLTZ and treated 0x02 - // / 0x12 as BGEZ — diverging from psxe. + // REGIMM default: branch type is decided by bit 16 + // (= rt & 1). Even-numbered rt → BLTZ semantics, odd → BGEZ. + // (An earlier version only matched 0x00/0x10 for BLTZ and treated + // 0x02 / 0x12 as BGEZ, which was wrong.) var take: bool = false; if ((variant & 1) == 0) { take = signedLtZero(s); @@ -797,7 +797,7 @@ pub fn iSltu(c: mut Cpu, opc: u32, regs: *mut[] u32) { pub fn iMfc0(c: mut Cpu, opc: u32, regs: *mut[] u32, cop0: *mut[] u32) { const v: u32 = cop0Read(cop0, opRd(opc)); - // psxe's mfc0 does DO_PENDING_LOAD unconditionally (cpu.c:1209) — + // mfc0 commits the pending load unconditionally — // a prior load to this same rt commits (briefly visible) before the // new delayed load is queued. The loads' `if loadD != rt` guard is // a different hazard; mfc* must not share it. @@ -806,8 +806,7 @@ pub fn iMfc0(c: mut Cpu, opc: u32, regs: *mut[] u32, cop0: *mut[] u32) { c.loadV = v; } -// COP0 mtc0 write mask, per cop0 register index. Mirrors psxe's -// g_psx_cpu_cop0_write_mask_table (cpu.c:10-27): CAUSE (r13) is mostly +// COP0 mtc0 write mask, per cop0 register index. CAUSE (r13) is mostly // read-only — only the software-IRQ bits 8-9 are writable, bit 10 // (IP2, the IC-driven external IRQ) MUST be preserved. Without this, // a kernel ISR that clears software IRQs by writing CAUSE clobbers @@ -849,14 +848,14 @@ pub fn iCop2Unimpl(c: mut Cpu, opc: u32, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32) { const rs: u32 = opRs(opc); if (rs == 0x00) { - // MFC2 — unconditional pending-load commit, like psxe (cpu.c:1449). + // MFC2 — unconditional pending-load commit. commitLoad(c, regs); c.loadD = opRt(opc); c.loadV = gteDataRead(bus.gte.ptr, opRd(opc)); return; } if (rs == 0x02) { - // CFC2 — unconditional pending-load commit, like psxe (cpu.c:1458). + // CFC2 — unconditional pending-load commit. commitLoad(c, regs); c.loadD = opRt(opc); c.loadV = gteCtrlRead(bus.gte.ptr, opRd(opc)); @@ -879,14 +878,14 @@ pub fn iCop2Unimpl(c: mut Cpu, opc: u32, bus: Bus, commitLoad(c, regs); gteExec(bus.gte.ptr, opc); // GTE math ops cost more than the 2-cycle base step() already charged - // (RTPS=15, NCDT=44, ...); add the remainder so device timing matches - // psxe through GTE-heavy code (cpu.c psx_cpu_execute GTE switch). + // (RTPS=15, NCDT=44, ...); add the remainder so device timing stays + // accurate through GTE-heavy code. c.cycles = c.cycles + (gteOpCycles(opc) as u64) - 2; } // LWC2 / SWC2 (primary 0x32 / 0x3A) — load/store a single COP2 data // register from/to memory. Used by some BIOS routines around RTPS -// calls. Per psxe these go through the standard bus + GTE data path. +// calls. These go through the standard bus + GTE data path. pub fn iCop2Lwc(c: mut Cpu, opc: u32, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32) { const s: u32 = gprRead(regs, opRs(opc)); @@ -910,9 +909,9 @@ pub fn iCop2Swc(c: mut Cpu, opc: u32, bus: Bus, raiseException(c, cop0, CAUSE_ADES); return; } - // Match psxe cpu.c psx_cpu_i_swc2: drop the write while the cache - // is isolated. Real R3000 routes the access to the I-cache instead - // of memory; emulators just no-op the store. + // Drop the write while the cache is isolated. Real R3000 routes the + // access to the I-cache instead of memory; emulators just no-op the + // store. if ((cop0Read(cop0, C0_SR) & SR_ISC) != 0) { return; } busWrite32(bus, cop0, addr, gteDataRead(bus.gte.ptr, opRt(opc))); } @@ -959,16 +958,16 @@ pub fn dispatchSpecial(c: mut Cpu, opc: u32, bus: Bus, } pub fn dispatchCop0(c: mut Cpu, opc: u32, regs: *mut[] u32, cop0: *mut[] u32) { - // psxe dispatches purely on rs (the CO/CT/MFC/MTC selector); the + // Dispatch purely on rs (the CO/CT/MFC/MTC selector); the // R3000 fn field is ignored for RFE since TLB ops don't exist on - // PSX. Match psxe: rs=0x10 → RFE regardless of fn. + // PSX. rs=0x10 → RFE regardless of fn. const rs: u32 = opRs(opc); if (rs == 0x00) { iMfc0(c, opc, regs, cop0); return; } if (rs == 0x04) { iMtc0(c, opc, regs, cop0); return; } if (rs == 0x10) { iRfe(c, regs, cop0); return; } // Unknown COP0 rs — real R3000 silently treats this as a NOP - // (verified via ps1-tests cpu/cop testCop0InvalidOpcode). psxe - // raises CAUSE_RI here, which is more aggressive than hardware. + // (verified via ps1-tests cpu/cop testCop0InvalidOpcode). Raising + // CAUSE_RI here would be more aggressive than hardware. commitLoad(c, regs); } @@ -1170,19 +1169,19 @@ pub fn step(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32) { } // Take an external interrupt before issuing the next instruction — - // matches psxe's check at the top of psx_cpu_cycle, after fetch and + // checked here after fetch and // before dispatch. CAUSE_INT has code 0 (already in CAUSE bits 6..2 // when raised), so we just trip the exception vector. // - // psxe `return`s after the exception so the current cycle does NOT + // `return` after the exception so the current cycle does NOT // dispatch the vector's first instruction — that happens on the - // next cycle. We do the same here; without the return, every IRQ - // handler runs 1 instruction ahead of psxe. + // next cycle. Without the return, every IRQ + // handler would run 1 instruction ahead. // - // Fetch-access cost (psxe cpu.c:302, last_cycles = bus access cycles): - // reading an opcode from the BIOS ROM costs 18 cycles (bios.c - // bus_delay=18); RAM/scratchpad/IO cost 0. This is the dominant + // Fetch-access cost (bus access cycles): + // reading an opcode from the BIOS ROM costs 18 cycles (BIOS bus + // delay=18); RAM/scratchpad/IO cost 0. This is the dominant // per-instruction timing difference from a flat model — BIOS code, // where the CD/IO polling loops live, advances device clocks ~10× // faster per instruction. @@ -1190,23 +1189,22 @@ pub fn step(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32) { if (inBios) { fetchCyc = 18; } if (cpuIrqPending(cop0)) { - // psxe quirk: a COP2 (GTE) math op "wins" over a pending IRQ — + // Hardware quirk: a COP2 (GTE) math op "wins" over a pending IRQ — // it executes before the interrupt is taken, and because EPC - // points back at it, it re-runs after the handler returns - // (cpu.c:323-358). Match by executing the op here when the + // points back at it, it re-runs after the handler returns. + // Handle this by executing the op here when the // instruction about to run (at c.pc == savedPc) is a GTE math // op: primary 0x12 with the COP2 command bit set (0x4A......). const irqOpc: u32 = busRead32(bus, c.pc); - // psxe charges fetch + (GTE op cycles if the op won); execute() does - // not run on an IRQ-taken cycle, so there is no +2 base here - // (cpu.c psx_cpu_cycle: total_cycles += last_cycles, then return). + // Charge fetch + (GTE op cycles if the op won); execute() does + // not run on an IRQ-taken cycle, so there is no +2 base here. var irqCyc: u64 = fetchCyc; if ((irqOpc & 0xFE000000) == 0x4A000000) { - // psxe runs DO_PENDING_LOAD as the first statement of this - // GTE-wins branch (cpu.c:369-371), and duckstation flushes the - // load on every exception. Commit the pending delayed load so the - // handler observes the loaded value (without this it commits one - // instruction late, at the handler's first instruction). + // Commit the pending delayed load as the first statement of this + // GTE-wins branch (DuckStation flushes the load on every + // exception) so the handler observes the loaded value (without + // this it commits one instruction late, at the handler's first + // instruction). commitLoad(c, regs); gteExec(bus.gte.ptr, irqOpc); irqCyc = irqCyc + (gteOpCycles(irqOpc) as u64); @@ -1222,8 +1220,8 @@ pub fn step(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32) { c.pc = c.nextPc; c.nextPc = c.nextPc + 4; - // psxe: last_cycles = fetch cycles + instruction cycles. Most - // instructions cost 2 (psx_cpu_execute returns 2); the COP2 math path + // Total cost = fetch cycles + instruction cycles. Most + // instructions cost 2; the COP2 math path // below adds the remainder of its GTE op cost on top of this base. c.cycles = c.cycles + fetchCyc + 2; diff --git a/cue.jam b/cue.jam index 889ac7a..11a9d51 100644 --- a/cue.jam +++ b/cue.jam @@ -1,4 +1,4 @@ -// CUE-sheet parser — port of psxe/psx/dev/cdrom/cue.c. +// CUE-sheet parser. // // A .cue text file describes how one or more .bin files map to CD-ROM // tracks. The minimum we need to parse is: @@ -13,7 +13,7 @@ // AUDIO, etc.) and one or two INDEX entries (00 = pregap start, 01 = // track data start, MSF format). // -// We use fixed-size tables (max 16 files, 99 tracks) instead of psxe's +// We use fixed-size tables (max 16 files, 99 tracks) instead of // linked lists — Jam's stdlib doesn't ship one and a CD can't have more // than 99 tracks anyway. Each track stores absolute LBA start/end so the // disc-read path can look up by LBA. @@ -23,7 +23,7 @@ const { File, exists, canonicalize } = import("std/fs"); const { Vec } = import("std/collections"); const { Box } = import("std/box"); -// Track-mode enum (psxe cue.h subset we actually use). +// Track-mode enum (subset we actually use). pub const TRACK_MODE_UNKNOWN: u32 = 0; pub const TRACK_MODE_AUDIO: u32 = 1; pub const TRACK_MODE_MODE1_2048: u32 = 2; @@ -72,7 +72,7 @@ pub const Cue = struct { pathBuf: Vec(u8), pathBufLen: u32, pathBufCap: u32, - // Parser scratch state (psxe-style readByte loop). + // Parser scratch state (readByte loop). file: File, cur: u32, // current character (sentinel-extended) rootPath: Vec(u8), // directory of the .cue, used to resolve BIN paths @@ -274,7 +274,7 @@ pub fn pathBufAppend(raw: *mut[] u8, src: *mut[] u8, len: u32) u32 { // track-range computation // // After parsing all FILE/TRACK/INDEX lines, walk each file's track list -// and compute startLba/endLba. Mirrors psxe init_tracks(). +// and compute startLba/endLba. pub fn cueComputeRanges(raw: *mut[] u8) { var c: *mut[] Cue = raw as *mut[] Cue; // Initial LBA is 150 (00:02:00 lead-in). @@ -364,8 +364,8 @@ pub fn cueGetTrackNumber(raw: *mut[] u8, lba: u32) u32 { return 1; } -// Locate the track owning `lba`. Returns trackCount on "past-end" (psxe's -// TS_FAR signal); a value < trackCount with mode=AUDIO means CDDA. +// Locate the track owning `lba`. Returns trackCount on "past-end"; a +// value < trackCount with mode=AUDIO means CDDA. pub fn cueFindTrack(raw: *mut[] u8, lba: u32) u32 { var c: *mut[] Cue = raw as *mut[] Cue; var i: u32 = 0; @@ -379,7 +379,7 @@ pub fn cueFindTrack(raw: *mut[] u8, lba: u32) u32 { } // Pregap sector layout: byte 0 = 0x00, bytes 1..10 = 0xFF (sync pattern), -// byte 11 = 0x00, the rest zero. Matches psxe cue.c:535-536. Local +// byte 11 = 0x00, the rest zero. Local // duplicate of disc.fillPregapSector — keeps cue.jam acyclic (disc // imports cue, not the other way around). fn fillPregap(buf: *mut[] u8) { @@ -399,7 +399,7 @@ pub fn cueRead(raw: *mut[] u8, lba: u32, buf: *mut[] u8) i32 { if (c[0].trackCount > 0 && lba >= c[0].tracks.ptr[c[0].trackCount - 1].endLba) { return 0; } - // Pregap: psxe cue.c:535-536 fills zero + sync at bytes [1..10]. + // Pregap: fill zero + sync at bytes [1..10]. fillPregap(buf); return 1; } @@ -432,7 +432,7 @@ pub fn cueQuery(raw: *mut[] u8, lba: u32) u32 { // parser // -// Stream the .cue text char-by-char (psxe's fgetc model). Each keyword +// Stream the .cue text char-by-char (fgetc model). Each keyword // dispatches to its handler; FILE pushes onto files[], TRACK pushes // onto tracks[] with the most recent file, INDEX records its MSF. pub fn cueParse(raw: *mut[] u8, path: []u8) i32 { diff --git a/disc.jam b/disc.jam index 8740249..d07a130 100644 --- a/disc.jam +++ b/disc.jam @@ -11,7 +11,7 @@ // audio/data type tagging. With raw .bin we treat the whole file as a // single Mode2/2352 data track. // -// LBA convention follows the PSX BIOS / psxe: +// LBA convention follows the PSX BIOS: // LBA = 150 → start of track 1 data (00:02:00 on the disc) // LBA = 150 + N → Nth sector of the data track in the .bin file // We subtract the 150-sector lead-in when seeking inside the .bin — @@ -131,7 +131,7 @@ pub fn discOpen(d: *mut[] Disc, path: []u8) i32 { // `buf`. Returns 1 on success, 0 on out-of-range or no-disc. // Fill a 2352-byte buffer with the PSX CD-ROM pregap sector layout: // byte 0 = 0x00, bytes 1..10 = 0xFF (sync pattern), byte 11 = 0x00, rest -// zero. Matches psxe cue.c:535-536. Some games inspect the sync pattern +// zero. Some games inspect the sync pattern // when reading pregap sectors and would otherwise see all-zero data. pub fn fillPregapSector(buf: *mut[] u8) { var i: u32 = 0; diff --git a/dma.jam b/dma.jam index bd2a5ff..17a521a 100644 --- a/dma.jam +++ b/dma.jam @@ -1,4 +1,4 @@ -// DMA controller (port of psxe/psx/dev/dma.c). +// DMA controller. // // Seven 32-bit channels share a 0x80-byte register file at 0x1F801080. // Channel writes that flip the CHCR busy/trigger bits kick off a @@ -7,7 +7,7 @@ // channels are stubbed enough to not stall the BIOS. // // dma.jam does NOT import bus.jam (that would form a cycle through -// bus.jam ↦ dma.jam ↦ bus.jam). Instead the Bus type is redeclared +// bus.jam -> dma.jam -> bus.jam). Instead the Bus type is redeclared // structurally and we reach into bus.ram.ptr / bus.gpu directly, calling // gpuGp0 / vram helpers from gpu.jam where necessary. @@ -48,8 +48,8 @@ const DICR_FLAGS: u32 = 0x7F000000; const DICR_IRQEN: u32 = 0x00800000; const DICR_IRQSI: u32 = 0x80000000; -// Owned DMA-state buffer. d[21] = DPCR is pre-set to the psxe default -// (psxe dma.c). Auto-drops with the Bus. +// Owned DMA-state buffer. d[21] = DPCR is pre-set to its reset default +// (0x07654321). Auto-drops with the Bus. pub fn dmaAlloc() Vec(u32) { var b: Vec(u32) = Vec(u32).filled(0, DMA_STATE_WORDS); b[21] = 0x07654321; @@ -92,7 +92,7 @@ pub fn ramWrite16(bus: Bus, addr: u32, val: u32) { // IO surface -// Per-channel CHCR hardwiring masks (psxe dma.c lines 14-24). Channels +// Per-channel CHCR hardwiring masks. Channels // 0..5 expose the full 32-bit CHCR through these mask bits; channel 6 // (OTC) hardwires bit 1 to 1 and only allows bits 24, 28, 30 to be set // via writes — everything else reads back as zero. ps1-tests dma/otc-test @@ -128,8 +128,8 @@ pub fn dmaRead32(d: *mut[] u32, off: u32) u32 { } // Sub-offset byte/halfword reads must extract from the aligned 32-bit -// register, mirroring psxe psx_dma_read16/read8 (which special-case 0x75/0x76 -// = DICR bytes). The old `dmaRead32(off) & mask` returned 0 for any offset not +// register, special-casing 0x75/0x76 +// = DICR bytes. The old `dmaRead32(off) & mask` returned 0 for any offset not // 0x70/0x74 (dmaRead32's default) — so a `lbu`/`lhu` of the DICR enable bits // at 0x1F8010F5/F6 read 0, and Brave Fencer's per-frame DICR-enable RMW then // wrote the per-channel IRQ enables back as 0 → DMA-completion IRQs stopped @@ -146,9 +146,9 @@ pub fn dmaRead8(d: *mut[] u32, off: u32) u32 { } pub fn dmaWriteDicr(d: *mut[] u32, val: u32) { - // Mirror psxe dma_write_dicr (dma.c:108-118) exactly. Preserve only - // bit 31 (IRQSI master flag) from the prior state, write-1-to-clear - // the channel flags, and copy the rest of the new value verbatim. + // Preserve only bit 31 (IRQSI master flag) from the prior state, + // write-1-to-clear the channel flags, and copy the rest of the new + // value verbatim. // Do NOT pre-recompute IRQSI here — dmaUpdate's edge detector // requires `prev = old IRQSI` to fire a 0->1 transition. const ack: u32 = val & DICR_FLAGS; @@ -205,7 +205,7 @@ pub fn dmaWrite8(d: *mut[] u32, bus: Bus, cop0: *mut[] u32, off: u32, val: u32) // Trigger a channel's transfer when CHCR is written with the busy bit // set. After the burst completes we immediately call dmaUpdate to fold // the per-channel d[23..28] "done" flags into DICR.FLAG bits and raise -// IC_DMA if enabled — psxe does this every CPU cycle via psx_dma_update, +// IC_DMA if enabled — this folding ideally happens every CPU cycle, // but our scanline-granularity update missed the window where the game // polls DICR right after CHCR (Brave Fencer's audio bank loader does // exactly this around 0x80042510). @@ -219,13 +219,13 @@ pub fn dmaTrigger(d: *mut[] u32, bus: Bus, cop0: *mut[] u32, ch: u32) { 6 { dmaDoOtc(d, bus); } _ {} } - // No store-time dmaUpdate: psxe's psx_dma_write32 only runs the transfer - // and defers all IRQ folding to the per-instruction psx_dma_update. The + // No store-time dmaUpdate: the CHCR write only runs the transfer and + // defers all IRQ folding to the per-instruction dmaUpdate. The // old store-time call here was a workaround for the obsolete per-scanline // update granularity; with per-instruction dmaUpdate (main.jam) it's // redundant and caused DMA IRQs to fire re-entrantly mid-store, ahead of - // psxe. The next instruction's dmaUpdate folds the done-flag in the same - // iteration psxe would (DICR polls one instruction later still see it). + // the per-instruction update. The next instruction's dmaUpdate folds the + // done-flag one instruction later, which DICR polls still see. } pub fn chcrBusy(d: *mut[] u32, ch: u32) bool { @@ -280,9 +280,9 @@ pub fn dmaDoGpuRequest(d: *mut[] u32, bus: Bus) { } } else { // GPU→RAM (C0 transfer): drain GPUREAD per word into RAM. - // psxe dma.c:281-289: `data = psx_bus_read32(bus, 0x1F801810)` — - // GPUREAD pulls from VRAM via the active C0 state machine - // (gpu.c:69-114). Our previous stub wrote zeros, which broke + // Each word reads GPUREAD at 0x1F801810, which + // pulls from VRAM via the active C0 state machine. + // Our previous stub wrote zeros, which broke // games that DMA-copy VRAM back to RAM (font caches, frame // composites, etc.). var i: u32 = 0; @@ -321,7 +321,7 @@ pub fn dmaDoGpuLinked(d: *mut[] u32, bus: Bus) { // data port (0x1F801820), which buffers it as command parameters or // command body. When all words for the active MDEC command are // delivered, MDEC's internal state machine runs the decode and fills -// its output buffer (mdec.jam:mdecWrite32). Matches psxe dma.c:179-205. +// its output buffer (mdec.jam:mdecWrite32). pub fn dmaDoMdecIn(d: *mut[] u32, bus: Bus) { if (!chcrBusy(d, 0)) { return; } const size: u32 = bcrSize(d, 0) * bcrBlocks(d, 0); @@ -340,7 +340,7 @@ pub fn dmaDoMdecIn(d: *mut[] u32, bus: Bus) { d[23] = size; } -// DMA1: MDEC output port → RAM. Match psxe exactly (dma.c:208-235). +// DMA1: MDEC output port → RAM. pub fn dmaDoMdecOut(d: *mut[] u32, bus: Bus) { if (!chcrBusy(d, 1)) { return; } const size: u32 = bcrSize(d, 1) * bcrBlocks(d, 1); @@ -360,8 +360,8 @@ pub fn dmaDoMdecOut(d: *mut[] u32, bus: Bus) { } // CD-ROM → RAM transfer. Each "word" in BCR.size is 4 bytes from the -// data FIFO at 0x1F801802. psxe streams via psx_cdrom_read32 which is -// our cdromReadData() popping one byte at a time and packing into a +// data FIFO at 0x1F801802. Streaming goes through +// cdromReadData(), popping one byte at a time and packing into a // little-endian word. After the transfer, ack via the d[26] flag so // dmaUpdate raises the channel-3 IRQ if enabled. pub fn dmaDoCdrom(d: *mut[] u32, bus: Bus) { @@ -369,10 +369,11 @@ pub fn dmaDoCdrom(d: *mut[] u32, bus: Bus) { // CD->RAM is SyncMode 0 (burst: BCR low 16 = word count) or SyncMode 1 // (block: low = block size in words, high = block count; total = // size*blocks). The old code used only bcrSize, so a SyncMode-1 read - // transferred ONE block instead of all of them — matching psxe but - // diverging from all four reference emulators (and the SPU channel here, - // which already multiplies). Under-reading made the game re-kick DMA3 per - // block, draining a partially-consumed FIFO into zero words. + // transferred ONE block instead of all of them — matching the original + // port but diverging from all four reference emulators (and the SPU + // channel here, which already multiplies). Under-reading made the game + // re-kick DMA3 per block, draining a partially-consumed FIFO into zero + // words. var size: u32 = 0; match (chcrSync(d, 3)) { 0 { size = bcrSize(d, 3); if (size == 0) { size = 0x10000; } } @@ -398,7 +399,7 @@ pub fn dmaDoCdrom(d: *mut[] u32, bus: Bus) { d[26] = 1; } -// SPU DMA (channel 4). psxe dma.c:375-437 pipes each 32-bit word through +// SPU DMA (channel 4). Each 32-bit word is piped through // the SPU TFIFO register at 0x1F801DA8 as two 16-bit halves, low first. // Reverse direction reads two halves from TFIFO and packs them. Previously // we drop the data — meaning Musashi/HM ADPCM banks were never reaching @@ -441,8 +442,8 @@ pub fn dmaDoSpu(d: *mut[] u32, bus: Bus) { } } d[chBase(4) + F_MADR] = addr; - // psxe psx_dma_do_spu completes the transfer instantly, FULLY clears CHCR - // (BUSY off), and raises the ch4 completion IRQ on the NEXT psx_dma_update + // The SPU transfer completes instantly, FULLY clears CHCR + // (BUSY off), and raises the ch4 completion IRQ on the NEXT dmaUpdate // (spu_irq_delay is set then immediately zeroed → IC_DMA ~1 instr later). // The old jam model kept BUSY for `words*8` cycles (d[F_SPU_REMAIN]) and // only THEN raised the IRQ — an in-flight-DMA nicety for ps1-tests @@ -451,9 +452,8 @@ pub fn dmaDoSpu(d: *mut[] u32, bus: Bus) { // 0x01000201) then polls I_STAT a few instructions later expecting that // IC_DMA; the deferral made jam miss the poll, so the game never dispatched // its DMA-IRQ handler — which is what advances the FMV's STR frame clock - // (0x800d24d0) — and the video stayed black. (Trace-diff vs psxe-strict - // pinned this as the FIRST jam<->psxe divergence, instr 280,834,781.) - // Match psxe: clear CHCR now + raise the ch4 done flag immediately so + // (0x800d24d0) — and the video stayed black. + // The fix: clear CHCR now + raise the ch4 done flag immediately so // dmaUpdate folds it into DICR.DMA4FL → IC_DMA on the next instruction. d[chBase(4) + F_CHCR] = 0; d[chBase(4) + F_BCR] = 0; @@ -500,14 +500,14 @@ pub fn dmaTickSpu(d: *mut[] u32, delta: u32) { } } -// Called once per CPU instruction (main.jam), exactly like psxe's -// psx_dma_update (psx.c:99). MDEC-in/out (ch0/ch1) are the only channels -// psxe genuinely DELAYS: it decrements mdec_in/out_irq_delay by 1 per call -// and raises the completion flag only when it reaches 0 — so the MDEC -// DMA-done IRQ fires `size` instructions AFTER the transfer (dma.c:525-539). +// Called once per CPU instruction (main.jam). MDEC-in/out (ch0/ch1) are +// the only channels that genuinely DELAY: the mdec-in/out irq-delay +// decrements by 1 per call and raises the completion flag only when it +// reaches 0 — so the MDEC DMA-done IRQ fires `size` instructions AFTER +// the transfer. // d[23]/d[24] hold that remaining-instruction COUNT (set to `size` by the -// transfer). All other channels fire on the next update (psxe zeroes their -// delay before the !delay test = immediate). Brave Fencer's FMV chains the +// transfer). All other channels fire on the next update (their delay is +// zeroed before the !delay test = immediate). Brave Fencer's FMV chains the // next MDEC frame-decode off this IRQ; jam's old immediate fire landed it // before the handler was ready, so the decode pipeline never re-fed the // MDEC → black video. diff --git a/gpu.jam b/gpu.jam index 7cfa0e7..331bd42 100644 --- a/gpu.jam +++ b/gpu.jam @@ -1,4 +1,4 @@ -// GPU (partial port of psxe/psx/dev/gpu.c). +// GPU (partial implementation). // // Implements the GP0/GP1 command machinery and the polygon, line, // rectangle, fill, and CPU↔VRAM blit commands. Triangles are @@ -25,7 +25,7 @@ // [46] gpustat // [47] display_mode // [48..51] draw_x1, draw_y1, draw_x2, draw_y2 -// [52..53] off_x, off_y (sign-bit set per psxe) +// [52..53] off_x, off_y (11-bit signed, sign-extended) // [54..57] texw_mx, texw_my, texw_ox, texw_oy // [58..59] clut_x, clut_y // [60..62] texp_x, texp_y, texp_d @@ -116,7 +116,7 @@ pub fn gpuRead32(g: *mut[] u32, vram: *mut[] u8, off: u32) u32 { return data; } if (off == 0x04) { - // psxe ORs GPUSTAT with 0x1C000000 to keep BIOS happy. + // OR GPUSTAT with 0x1C000000 to keep the BIOS happy. return g[46] | 0x1C000000; } return 0; @@ -199,7 +199,7 @@ pub fn gpuGp0(g: *mut[] u32, vram: *mut[] u8, val: u32) { // GP1 (control) — display enable, DMA direction, display area, etc. pub fn gpuGp1(g: *mut[] u32, val: u32) { match ((val >> 24) & 0xFF) { - // GP1(0x00): psxe ignores it (gpu.c:1841-1874 switch has no case 0). + // GP1(0x00): leave gpustat untouched here. // A previous clobber of gpustat wiped texpage / display-enable bits // accumulated from prior GP0 commands; diff-trace showed Musashi // reading GPUSTAT and getting the wrong value at PC=0x8005C204. @@ -212,18 +212,16 @@ pub fn gpuGp1(g: *mut[] u32, val: u32) { } // Acknowledge GPU IRQ (clear bit 24 of GPUSTAT). 0x02 { g[46] = g[46] & (~0x01000000); } - // Display enable: bit 23 of GPUSTAT = 1 when disabled (psxe). + // Display enable: bit 23 of GPUSTAT = 1 when disabled. 0x03 { g[46] = g[46] & (~0x00800000); g[46] = g[46] | ((val << 23) & 0x00800000); } - // GP1(04h) DMA direction. psxe IGNORES this command entirely - // (gpu.c:1850 `case 0x04: {} break;`) — it never reflects the - // direction into GPUSTAT bits 29-30, so they stay 0. Real hardware - // *does* set them, but matching psxe is the goal: storing the - // direction here made GPUSTAT bit 30 read 1 (dir 2) where psxe - // reads 0, and a game branch on that bit diverged the two - // (REG_TRACE cmp, L≈19.477M). Keep it a no-op. + // GP1(04h) DMA direction. Handled as a no-op here: the direction is + // never reflected into GPUSTAT bits 29-30, so they stay 0. Real + // hardware *does* set them, but reflecting the direction here made + // GPUSTAT bit 30 read 1 (dir 2), and a game branch on that bit then + // diverged (REG_TRACE cmp, L≈19.477M). Keep it a no-op. 0x04 { } 0x05 { g[63] = val & 0x3FF; @@ -250,7 +248,7 @@ pub fn gpuGp1(g: *mut[] u32, val: u32) { // GP0 command parser — branches on the top three bits of buf[0] for // the polygon / line / rect families, and falls through to a per-byte -// switch for everything else. Mirrors psxe gpu.c psx_gpu_update_cmd. +// switch for everything else. pub fn gpuUpdateCmd(g: *mut[] u32, vram: *mut[] u8) { const cmd: u32 = (g[0] >> 24) & 0xFF; const family: u32 = (g[0] >> 29) & 7; @@ -343,7 +341,7 @@ pub fn gpuUpdateCmd(g: *mut[] u32, vram: *mut[] u8) { } // 0x02 fill rect in VRAM. The colour is taken straight from buf[0] -// (no dither), and the area is masked to a 16-pixel grid per psxe. +// (no dither), and the area is masked to a 16-pixel grid. pub fn gpuFillRect(g: *mut[] u32, vram: *mut[] u8) { if (g[19] == GP_RECV_CMD) { g[19] = GP_RECV_ARGS; @@ -368,7 +366,7 @@ pub fn gpuFillRect(g: *mut[] u32, vram: *mut[] u8) { g[19] = GP_RECV_CMD; } -// 0x80 VRAM-to-VRAM copy (psxe gpu_cmd_80). +// 0x80 VRAM-to-VRAM copy. pub fn gpuVramToVram(g: *mut[] u32, vram: *mut[] u8) { if (g[19] == GP_RECV_CMD) { g[19] = GP_RECV_ARGS; @@ -419,13 +417,13 @@ pub fn gpuCpuToVram(g: *mut[] u32, vram: *mut[] u8) { g[19] = GP_RECV_DATA; return; } - // RECV_DATA — one 32-bit word delivers two VRAM pixels. psxe's - // gpu_cmd_a0 (gpu.c:1122-1137) writes the destination directly, - // bypassing the GP0(0xE6) mask bits — same as all of its other - // draw commands. Mirror that here so Musashi's FMV (which leaves - // check-mask set when uploading decoded frames) doesn't silently - // drop writes to pixels that previously had bit 15 set. The strict - // spec says masks SHOULD apply, but no real game relies on it. + // RECV_DATA — one 32-bit word delivers two VRAM pixels. GP0(0xA0) + // writes the destination directly, bypassing the GP0(0xE6) mask + // bits — same as all the other draw commands here. This is done so + // Musashi's FMV (which leaves check-mask set when uploading decoded + // frames) doesn't silently drop writes to pixels that previously had + // bit 15 set. The strict spec says masks SHOULD apply, but no real + // game relies on it. const x0: u32 = (g[22] + g[28]) & 0x3FF; const y0: u32 = (g[23] + g[29]) & 0x1FF; vramWritePixel(vram, x0, y0, g[20] & 0xFFFF); @@ -617,7 +615,6 @@ pub fn gpuRasterPoly(g: *mut[] u32, vram: *mut[] u8) { // tex flat buf[0]=cmd+col, then per-vert (xy, page/clut+uv) // mono shaded buf[0]=col0+cmd, then per-vert (xy, colN+pad) // tex shaded buf[0]=col0+cmd, then per-vert (xy, page/clut+uv, colN) - // The shaded textured layout matches psxe gpu_cmd_3c. if (isTex && isGour) { // texture + gouraud (cmd 0x34, 0x3C, 0x3E) v0x = gpuVertX(g[1]); v0y = gpuVertY(g[1]); @@ -632,11 +629,10 @@ pub fn gpuRasterPoly(g: *mut[] u32, vram: *mut[] u8) { tpx = (page & 0xF) << 6; tpy = (page & 0x10) << 4; depth = (page >> 7) & 3; - // psxe gpu.c:971-977 "Undocumented behavior — fixes Mortal Kombat - // II, Bubble Bobble, Driver 1 & 2". Textured polys update gpustat - // bits 0-8 from the polygon's tpage byte. Without this, GPUSTAT - // texpage bits drift from psxe's value (diff trace at instruction - // 470014 confirmed this is the source). + // Undocumented behaviour (fixes Mortal Kombat II, Bubble Bobble, + // Driver 1 & 2): textured polys update GPUSTAT bits 0-8 from the + // polygon's tpage byte. Without this, the GPUSTAT texpage bits + // drift from the expected value. // Undocumented behaviour: textured polys also update GPUSTAT // bits 0-8 from the tpage byte, and (when GP1(0x09) "allow // texture disable" is on) also drive bit 15 from tpage bit 11. @@ -644,8 +640,8 @@ pub fn gpuRasterPoly(g: *mut[] u32, vram: *mut[] u8) { g[46] = (g[46] & 0xFFFFFE00) | (page & 0x1FF); g[70] = (page >> 11) & 1; g[46] = (g[46] & 0xFFFF7FFF) | ((g[71] & g[70]) << 15); - // psxe gpu.c:972-974 — the polygon's tpage byte also rewrites - // the GLOBAL texp_x/y/d (the derived fields textured rectangles + // The polygon's tpage byte also rewrites the GLOBAL texp_x/y/d + // (the derived fields textured rectangles // read from). Without this, FMVs that draw the decoded frame // as a texture-mapped polygon then composite it with a // textured rect end up sampling the wrong VRAM region for the @@ -681,9 +677,9 @@ pub fn gpuRasterPoly(g: *mut[] u32, vram: *mut[] u8) { g[46] = (g[46] & 0xFFFFFE00) | (page & 0x1FF); g[70] = (page >> 11) & 1; g[46] = (g[46] & 0xFFFF7FFF) | ((g[71] & g[70]) << 15); - // psxe gpu.c:969-977 "Undocumented behavior? Fixes Mortal - // Kombat II, Bubble Bobble, Driver 1 & 2" — and Musashi FMV. - // Propagate the polygon's tpage into the derived globals so a + // Undocumented behaviour (fixes Mortal Kombat II, Bubble Bobble, + // Driver 1 & 2 — and Musashi FMV): + // propagate the polygon's tpage into the derived globals so a // following textured RECTANGLE (which carries clut but no // tpage) inherits this state instead of stale values. Without // it, the rect samples the wrong VRAM region → texel 0 → @@ -720,12 +716,12 @@ pub fn gpuRasterPoly(g: *mut[] u32, vram: *mut[] u8) { const baseColor: u32 = g[0] & 0xFFFFFF; - // Per psxe: textured polys read transparency mode from the texture - // page register's bits 5-6 (overrides GPUSTAT for this primitive); + // Textured polys read transparency mode from the texture page + // register's bits 5-6 (overrides GPUSTAT for this primitive); // untextured / shaded polys read it from GPUSTAT bits 5-6. const isTransp: bool = (flags & 0x02) != 0; var transpMode: u32 = (g[46] >> 5) & 3; - // psxe uses the texture-page transp mode for ALL textured polys, not + // Use the texture-page transp mode for ALL textured polys, not // just flat ones. The tpage word sits at buf[4] for flat-textured and // at buf[5] for shaded-textured (the extra per-vertex colour words // shift it), so pick the right word by isGour. @@ -770,7 +766,7 @@ pub fn gpuRasterTri(g: *mut[] u32, vram: *mut[] u8, var cx: u32 = cx0 + ox; var cy: u32 = cy0 + oy; // UV and per-vertex colour stay paired with their vertex when we - // flip winding (per psxe). + // flip winding. var bu2: u32 = bu; var bv2: u32 = bv; var cu2: u32 = cu; var cv2: u32 = cv; var bc2: u32 = bc; @@ -793,8 +789,7 @@ pub fn gpuRasterTri(g: *mut[] u32, vram: *mut[] u8, const xmax: u32 = gpuMax3(ax, bx, cx); const ymax: u32 = gpuMax3(ay, by, cy); // Reject only triangles wider/taller than the GPU's real limits - // (psxe gpu.c:265 uses 2048 / 1024). The earlier 1024/512 dropped - // valid large polygons. + // (2048 / 1024). The earlier 1024/512 dropped valid large polygons. if (sext32(xmax - xmin) > 2048 || sext32(ymax - ymin) > 1024) { return; } const absArea: i64 = gpuAbsI64(gpuEdge(ax, ay, bx, by, cx, cy)); @@ -841,9 +836,8 @@ pub fn gpuRasterTri(g: *mut[] u32, vram: *mut[] u8, color = texel; } else { // Modulate by the per-vertex gouraud colour - // when shaded (psxe gpu_render_triangle uses - // the interpolated colour as the modulator, - // gpu.c:299-327,348-354), else the flat base. + // when shaded (the interpolated colour is the + // modulator), else the flat base. var modc: u32 = baseColor; if (isGour) { const ar: i64 = (ac as i64) & 0xFF; @@ -876,10 +870,9 @@ pub fn gpuRasterTri(g: *mut[] u32, vram: *mut[] u8, } else { // Untextured: gouraud-interpolate the three // per-vertex colours by barycentric weights, or - // use the base colour if not shaded. Matches - // psxe `gpu_render_triangle` shaded path, with - // the same 4x4 ordered dither kernel applied - // pre-clamp for gradient banding suppression. + // use the base colour if not shaded. A 4x4 ordered + // dither kernel is applied pre-clamp for gradient + // banding suppression. var rgb: u32 = baseColor; if (isGour) { const ar: i64 = (ac as i64) & 0xFF; @@ -895,7 +888,7 @@ pub fn gpuRasterTri(g: *mut[] u32, vram: *mut[] u8, const nr: i64 = z0 * ar + z1 * br + z2 * cr; const ng: i64 = z0 * ag + z1 * bg + z2 * cg; const nb: i64 = z0 * ab + z1 * bb + z2 * cb; - // psxe dither offset from a 4x4 kernel indexed by + // dither offset from a 4x4 kernel indexed by // (x - xmin) & 3 and (y - ymin) & 3. const dx: u32 = (x - xmin) & 3; const dy: u32 = (y - ymin) & 3; @@ -926,7 +919,7 @@ pub fn gpuAbsI64(v: i64) i64 { return v; } -// 4x4 ordered dither kernel from psxe (`g_psx_gpu_dither_kernel`). +// 4x4 ordered dither kernel. // Indexed by `dy * 4 + dx` where `dx = (x - xmin) & 3`, `dy = (y - ymin) & 3`. // Values added to each interpolated channel before clamping reduce visible // banding on gouraud gradients (the BIOS's diamond logo uses this heavily). @@ -953,7 +946,7 @@ pub fn gpuDitherKernel(idx: u32) i64 { // Saturate one shaded channel: divide the barycentric numerator by the // triangle area, apply the dither offset, then clamp to [0, 0xFF]. -// Match psxe's order: divide first, dither, clamp. +// Order matters: divide first, dither, clamp. pub fn gpuSatChan(num: i64, area: i64, dith: i64) u32 { if (area == 0) { return 0; } var quot: i64 = num / area; @@ -981,11 +974,10 @@ pub fn gpuLineCmd(g: *mut[] u32, vram: *mut[] u8) { const isPoly: bool = (flags & 0x08) != 0; if (g[19] == GP_RECV_CMD) { g[19] = GP_RECV_ARGS; - // psxe gpu_line (gpu.c:1019-1031): poly-lines (bit 0x08) set - // cmd_args_remaining = -1 and drain words until a vertex word - // masks to 0x50005000; single segments take 2 verts (3 if - // gouraud). g[17] is unused on the poly-line path — it ends on - // the terminator below, not on an arg count. + // Poly-lines (bit 0x08) drain words until a vertex word masks to + // 0x50005000; single segments take 2 verts (3 if gouraud). g[17] + // is unused on the poly-line path — it ends on the terminator + // below, not on an arg count. if (isPoly) { g[17] = 2; return; } var args: u32 = 2; if (isGour) { args = args + 1; } @@ -994,13 +986,12 @@ pub fn gpuLineCmd(g: *mut[] u32, vram: *mut[] u8) { } // RECV_ARGS. if (isPoly) { - // Match psxe: a poly-line ends when the most-recently received - // word masks to 0x50005000, and it draws NOTHING (psxe's render - // loop is commented out, gpu.c:1033-1065). Bound the FIFO write - // index so we don't overflow the 16-word buffer — psxe lets - // buf_index grow (a latent overflow), but with no draw the words - // are never read back, so overwriting one slot is observably the - // same while keeping jam's state intact. + // A poly-line ends when the most-recently received word masks to + // 0x50005000, and it draws NOTHING (the poly-line render path is + // not implemented). Bound the FIFO write index so we don't + // overflow the 16-word buffer — with no draw the words are never + // read back, so overwriting one slot is observably the same while + // keeping jam's state intact. const last: u32 = g[g[16] - 1]; if ((last & 0xF000F000) == 0x50005000) { g[19] = GP_RECV_CMD; @@ -1028,8 +1019,7 @@ pub fn gpuLineCmd(g: *mut[] u32, vram: *mut[] u8) { g[19] = GP_RECV_CMD; } -// Bresenham line rasterizer. Mirrors psxe's plotLine / plotLineLow / -// plotLineHigh in src/dev/gpu.c — split on whether dx or dy dominates +// Bresenham line rasterizer — split on whether dx or dy dominates // (i.e. shallow vs. steep slope) and on direction, so the inner loop // always increments by 1 in the dominant axis and uses Bresenham's // error term to step in the minor axis. @@ -1156,16 +1146,15 @@ pub fn gpuRectCmd(g: *mut[] u32, vram: *mut[] u8) { } _ {} } - // psxe order: add the drawing offset first, *then* sign-extend + // Order matters: add the drawing offset first, *then* sign-extend // the 11-bit screen-space position. SE-after-offset is what the // hardware does — anything that overflows 11 bits wraps. var x: u32 = gpuSe11((xy & 0xFFFF) + g[52]); var y: u32 = gpuSe11(((xy >> 16) & 0xFFFF) + g[53]); // Drawing-area bounds. Pixels outside the (draw_x1..draw_x2, - // draw_y1..draw_y2) box are skipped; mirrors psxe's - // gpu_render_flat_rectangle which clamps the iteration limits - // against the drawing area. + // draw_y1..draw_y2) box are skipped: the iteration limits are + // clamped against the drawing area. const dx1: i64 = sext32(g[48]); const dy1: i64 = sext32(g[49]); const dx2: i64 = sext32(g[50]); @@ -1222,9 +1211,9 @@ pub fn gpuRectCmd(g: *mut[] u32, vram: *mut[] u8) { if (!isRaw) { color = gpuModulate(texel, g[0] & 0xFFFFFF); } - // Per psxe: when the rect's transparency flag is - // set, texel bit 15 selects whether THIS pixel - // blends. When the flag isn't set, no blending. + // When the rect's transparency flag is set, texel + // bit 15 selects whether THIS pixel blends. When + // the flag isn't set, no blending. var thisTransp: bool = false; if (isTransp) { thisTransp = (texel & 0x8000) != 0; @@ -1247,7 +1236,7 @@ pub fn gpuRectCmd(g: *mut[] u32, vram: *mut[] u8) { } // Apply the GP0(0xE2) texture window to a texel coordinate, then wrap to -// the 256x256 page. Mirrors psxe gpu_fetch_texel (gpu.c:150-153): +// the 256x256 page: // t = (t & ~mask) | (offset & mask); t &= 0xff // The window regs are pre-shifted to pixel units in g[54..57] // (mask_x, mask_y, off_x, off_y). With no window set (mask 0) this is a @@ -1293,8 +1282,8 @@ pub fn gpuModulate(texel: u32, mod: u32) u32 { const mr: u32 = mod & 0xFF; const mg: u32 = (mod >> 8) & 0xFF; const mb: u32 = (mod >> 16) & 0xFF; - // psxe rounds: roundf((tex * mod) / 128) (gpu.c:352-362). +0x40 - // before >>7 gives round-half-up, which matches for non-negative + // Modulation rounds to nearest: round((tex * mod) / 128). +0x40 + // before >>7 gives round-half-up, which is correct for non-negative // channels (tex, mod are unsigned here). var cr: u32 = ((tr * mr) + 0x40) >> 7; var cg: u32 = ((tg * mg) + 0x40) >> 7; @@ -1305,7 +1294,7 @@ pub fn gpuModulate(texel: u32, mod: u32) u32 { return gpuToBgr555(cr | (cg << 8) | (cb << 16)); } -// Semi-transparency blend (port of psxe's `if (transp)` block). +// Semi-transparency blend. // `fore` is a BGR555 foreground colour, `back` the existing VRAM pixel; // `mode` is the 2-bit semi-transparency mode out of GPUSTAT bits 5-6 // (or for textured polys, out of the texture-page register): @@ -1334,7 +1323,7 @@ pub fn gpuBlend(fore: u32, back: u32, mode: u32) u32 { cg = bg + fg; cb = bb + fb; } else if (mode == 2) { - // psxe uses `br - fr`; we'd underflow in u32, so clamp at 0. + // The blend is `br - fr`; we'd underflow in u32, so clamp at 0. if (br > fr) { cr = br - fr; } else { cr = 0; } if (bg > fg) { cg = bg - fg; } else { cg = 0; } if (bb > fb) { cb = bb - fb; } else { cb = 0; } @@ -1382,7 +1371,7 @@ pub const Gpu = struct { pub fn init() Self { var s: Self = Self { buf: [0; 80] }; s.buf[19] = GP_RECV_CMD; - // psxe init (gpu.c:59): only sets bit 23 (display disabled). + // Initial GPUSTAT: only bit 23 set (display disabled). // Reads OR 0x1C000000 on the fly. s.buf[46] = 0x00800000; s.buf[47] = 1; // default display_mode diff --git a/gte.jam b/gte.jam index b5a3588..625da09 100644 --- a/gte.jam +++ b/gte.jam @@ -1,18 +1,18 @@ // COP2 / GTE — Geometry Transformation Engine. // -// Port of psxe's GTE (cpu.c ~3000 lines of fixed-point math with -// saturation flags). This module covers the register file (32 data + +// PSX GTE — fixed-point geometry math with saturation flags. This +// module covers the register file (32 data + // 32 control regs), the register-move instructions (MFC2 / MTC2 / // CFC2 / CTC2 / LWC2 / SWC2), and the math ops dispatched by gteExec. // // Implemented ops: RTPS, RTPT (3xRTPS), NCLIP, OP, MVMVA, NCDS, NCDT, // SQR, AVSZ3, AVSZ4, DPCS, INTPL, CDP, NCCS, CC, NCS, NCT, DCPL, DPCT, -// GPF, GPL, NCCT — 22 of psxe's 22 GTE ops. +// GPF, GPL, NCCT — all 22 GTE ops. // -// Saturation FLAG bookkeeping is psxe-faithful: every clamp (IR/IR0/SXY/ +// Saturation FLAG bookkeeping is hardware-faithful: every clamp (IR/IR0/SXY/ // SZ3/RGB/MAC0) ORs its FLAG bit, the 44-bit MAC overflow is truncated -// per-term (gteCheckMac) and nested between additive terms exactly as -// psxe does, and bit 31 is synthesized from the error summary on read. +// per-term (gteCheckMac) and nested between additive terms, and bit 31 is +// synthesized from the error summary on read. // // Data register layout (32 × u32): // 0 V0 (xy packed, low half of vertex 0) @@ -63,9 +63,9 @@ const C_OFY: u32 = 25; const C_FLAG: u32 = 31; // Unsigned Newton-Raphson reciprocal table for the GTE perspective -// divide (psxe cpu.c:60-94, g_psx_gte_unr_table). 257 entries — the -// hardware uses this exact LUT, so a naive integer divide diverges from -// psxe/hardware on the projected SX/SY by up to a few units. +// divide. 257 entries — the hardware uses this exact LUT, so a naive +// integer divide diverges from hardware on the projected SX/SY by up to +// a few units. const GTE_UNR_TABLE: [257]u8 = [ 0xff, 0xfd, 0xfb, 0xf9, 0xf7, 0xf5, 0xf3, 0xf1, 0xef, 0xee, 0xec, 0xea, 0xe8, 0xe6, 0xe4, 0xe3, @@ -108,17 +108,16 @@ pub fn gteAlloc() Vec(u32) { return Vec(u32).filled(0, GTE_REGS); } // Data and control share the same storage buffer: data at index 0..31, // control at index 32..63. // -// Several COP2 data registers have side effects on read/write that -// psxe handles in gte_read_register / gte_write_register: +// Several COP2 data registers have side effects on read/write: // // 15 SXYP read → mirror of SXY2; write → push SXY FIFO // 28 IRGB write → unpack 15-bit RGB into IR1..IR3; read → repack // 29 ORGB read → repack IR1..IR3 to 15-bit; write → ignored (RO) // 30 LZCS write → also compute LZCR (leading sign-bit count) // 31 LZCR read-only -// Sign-extend a 16-bit value held in a u32 to full 32 bits. psxe stores -// several GTE registers as int16_t, so reading them back via MFC2/CFC2 -// sign-extends the low half (cpu.c gte_read_register). +// Sign-extend a 16-bit value held in a u32 to full 32 bits. Several GTE +// registers are stored as int16_t, so reading them back via MFC2/CFC2 +// sign-extends the low half. pub fn gteSext16(v: u32) u32 { if ((v & 0x8000) != 0) { return v | 0xFFFF0000; } return v & 0x0000FFFF; @@ -128,7 +127,7 @@ pub fn gteDataRead(g: *mut[] u32, idx: u32) u32 { const i: u32 = idx & 0x1F; if (i == 15) { return g[14]; } // SXYP mirrors SXY2 // 16-bit signed data registers: vZ (1/3/5) and IR0..IR3 (8..11) are - // int16_t in psxe, so MFC2 sign-extends the low half (cpu.c:1275-1285). + // int16_t, so MFC2 sign-extends the low half. if (i == 1 || i == 3 || i == 5 || i == 8 || i == 9 || i == 10 || i == 11) { return gteSext16(g[i]); @@ -137,8 +136,8 @@ pub fn gteDataRead(g: *mut[] u32, idx: u32) u32 { // Re-pack IR1/IR2/IR3 (>> 7, clamped 0..0x1F) into 15-bit RGB. // IR is stored zero-extended in the low 16 bits, so sign-extend // first (s16ToI64) — otherwise a negative IR reads as a large - // positive value and clamps to 0x1F instead of 0 (psxe cpu.c:1290, - // duckstation gte.cpp:343 both give 0 for IR<0). + // positive value and clamps to 0x1F instead of 0 (DuckStation + // gte.cpp:343 gives 0 for IR<0). var r: i64 = s16ToI64(g[9]) >> 7; var grn: i64 = s16ToI64(g[10]) >> 7; var b: i64 = s16ToI64(g[11]) >> 7; @@ -176,7 +175,7 @@ pub fn gteDataWrite(g: *mut[] u32, idx: u32, val: u32) { return; } if (i == 29) { - // ORGB is read-only — psxe silently discards writes. + // ORGB is read-only — writes are silently discarded. return; } if (i == 30) { @@ -209,16 +208,16 @@ pub fn gteCountLeadingSign(v: u32) u32 { pub fn gteCtrlRead(g: *mut[] u32, idx: u32) u32 { const i: u32 = idx & 0x1F; if (i == C_FLAG) { - // psxe cpu.c:1272-1273: synthesize "any-error" bit 31 from the - // overflow/saturation bit-mask 0x7F87E000, mask the low garbage. + // Synthesize "any-error" bit 31 from the overflow/saturation + // bit-mask 0x7F87E000, mask the low garbage. // Games CFC2 r63 and branch on bit 31; without this synthesis // they always see "no error" even when MAC/IR saturated. const v: u32 = g[32 + C_FLAG] & 0x7FFFF000; if ((v & 0x7F87E000) != 0) { return v | 0x80000000; } return v; } - // 16-bit signed control registers, sign-extended on read like psxe - // (cpu.c:1310,1318,1326,1332-1336): the three matrix m33 entries + // 16-bit signed control registers, sign-extended on read: the three + // matrix m33 entries // (rt/l/lr → idx 4/12/20), H (26), DQA (27), ZSF3 (29), ZSF4 (30). if (i == 4 || i == 12 || i == 20 || i == 26 || i == 27 || i == 29 || i == 30) { @@ -238,10 +237,9 @@ pub fn gteCtrlWrite(g: *mut[] u32, idx: u32, val: u32) { g[32 + i] = val; } -// Dispatch on the bottom 6 bits of the GTE math opcode. psxe's full -// table covers 22 ops (RTPS=0x01, NCLIP=0x06, ..., NCCT=0x3F) — all 22 -// are implemented below. FLAG is cleared at the top of every op (psxe -// sets R_FLAG = 0 in each psx_gte_i_*). +// Dispatch on the bottom 6 bits of the GTE math opcode. The full table +// covers 22 ops (RTPS=0x01, NCLIP=0x06, ..., NCCT=0x3F) — all 22 are +// implemented below. FLAG is cleared at the top of every op. pub fn gteExec(g: *mut[] u32, opc: u32) { g[32 + C_FLAG] = 0; // clear FLAG every instruction const op: u32 = opc & 0x3F; @@ -294,11 +292,10 @@ pub fn gteExec(g: *mut[] u32, opc: u32) { } // CPU-cycle cost of a COP2 (GTE) math op, keyed by the low 6 bits of the -// opcode. These mirror psxe's per-op cycle counts in the GTE switch -// (cpu.c psx_cpu_cycle / psx_cpu_execute) exactly. The CPU dispatcher adds -// this — instead of the flat 2-cycle base — when it runs a GTE math op, so -// device timing (DMA, CDROM, timers) advances at psxe's rate through -// GTE-heavy code. Unknown sub-ops fall back to the 2-cycle base. +// opcode. These are the hardware per-op cycle counts. The CPU dispatcher +// adds this — instead of the flat 2-cycle base — when it runs a GTE math +// op, so device timing (DMA, CDROM, timers) advances at the correct rate +// through GTE-heavy code. Unknown sub-ops fall back to the 2-cycle base. pub fn gteOpCycles(opc: u32) u32 { const op: u32 = opc & 0x3F; if (op == 0x01) { return 15; } // RTPS @@ -339,8 +336,8 @@ pub fn gtePushRgb(g: *mut[] u32, m1: i64, m2: i64, m3: i64) { } // Light-matrix transform of vertex vidx into (IR1,IR2,IR3) and (MAC1,MAC2,MAC3). -// Shared by NCS, NCDS, NCCS — psxe NCS/NCDS/NCCS macros all start with -// the same L*V transformation. Stores IR into g[9..11] and MAC into g[D_MAC*]. +// Shared by NCS, NCDS, NCCS — they all start with the same L*V +// transformation. Stores IR into g[9..11] and MAC into g[D_MAC*]. pub fn gteApplyLight(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { const vSlot: u32 = vidx * 2; const vxy: u32 = g[vSlot]; @@ -386,10 +383,10 @@ pub fn gteApplyLightColor(g: *mut[] u32, sf: u32, lm: u32) { const rbk: i64 = s32ToI64(g[32 + 13]); const gbk: i64 = s32ToI64(g[32 + 14]); const bbk: i64 = s32ToI64(g[32 + 15]); - // psxe nests gte_check_mac between the BK<<12 term and each LC*IR term - // so an intermediate exceeding 44 bits wraps before the next add - // (cpu.c:1800-1802). The L*V first stage is a single clamp_mac (see - // gteApplyLight) and is intentionally left unnested to match psxe. + // gteCheckMac nests between the BK<<12 term and each LC*IR term so an + // intermediate exceeding 44 bits wraps before the next add. The L*V + // first stage is a single gteClampMac (see gteApplyLight) and is + // intentionally left unnested, matching hardware. const a1a: i64 = gteCheckMac(g, 1, (rbk << 12) + lr1 * ir1); const a1b: i64 = gteCheckMac(g, 1, a1a + lr2 * ir2); const m1: i64 = gteClampMac(g, 1, a1b + lr3 * ir3, sf); @@ -448,8 +445,8 @@ pub fn gteNcs(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { const m2: i64 = s32ToI64(g[D_MAC2]); const m3: i64 = s32ToI64(g[D_MAC3]); gtePushRgb(g, m1, m2, m3); - // psxe NCS macro reclamps IR after the FIFO push using the *new* - // MAC values — the existing MAC1..MAC3 register state. + // NCS reclamps IR after the FIFO push using the *new* MAC values — + // the existing MAC1..MAC3 register state. gteIrTriple(g, m1, m2, m3, lm); } @@ -509,8 +506,8 @@ pub fn gteCdp(g: *mut[] u32, sf: u32, lm: u32) { gtePushRgb(g, m1, m2, m3); } -// DPCS — depth queue color single. FC interpolation on the RGBC color. -// psxe uses (RFC<<12) - (RC<<16), then (RC<<16) + IR0*ir'. The shift-by-16 +// DPCS — depth queue color single. FC interpolation on the RGBC color: +// (RFC<<12) - (RC<<16), then (RC<<16) + IR0*ir'. The shift-by-16 // difference vs FarColorInterp's (RC<<4)*IR is because here IR=1.0 implicit. pub fn gteDpcs(g: *mut[] u32, sf: u32, lm: u32) { const rgbc: u32 = g[6]; @@ -540,8 +537,8 @@ pub fn gteDpcs(g: *mut[] u32, sf: u32, lm: u32) { } // DPCT — depth queue color triple. Repeats DPCS three times, each time -// using RGB0 (FIFO head) as the source color. psxe macro DPCT1 also -// shifts the FIFO so each iteration's "RGB0" is the previous result. +// using RGB0 (FIFO head) as the source color. The FIFO shifts each +// iteration so the next "RGB0" is the previous result. pub fn gteDpct(g: *mut[] u32, sf: u32, lm: u32) { var iter: u32 = 0; while (iter < 3) { @@ -574,7 +571,7 @@ pub fn gteDpct(g: *mut[] u32, sf: u32, lm: u32) { } // INTPL — interpolation. FC interpolation using the *existing* IR as the -// "color" input (psxe: ir' = clamp(FC<<12 - IR<<12); MAC = IR<<12 + IR0*ir'). +// "color" input: ir' = clamp(FC<<12 - IR<<12); MAC = IR<<12 + IR0*ir'. pub fn gteIntpl(g: *mut[] u32, sf: u32, lm: u32) { const ir1: i64 = s16ToI64(g[9]); const ir2: i64 = s16ToI64(g[10]); @@ -676,10 +673,9 @@ pub fn s32ToI64(v: u32) i64 { return w; } -// Saturate an i64 to a signed 16-bit IR result. psxe's -// gte_clamp_ir(c, idx, value, lm) uses (lm ? 0 : -0x8000) as the lower -// bound. We default to lm=0 (signed 16) here; see `satIrLm` for the -// lm-aware variant used by RTPS/MVMVA. +// Saturate an i64 to a signed 16-bit IR result. The IR clamp uses +// (lm ? 0 : -0x8000) as the lower bound; this helper defaults to lm=0 +// (signed 16). See `satIrLm` for the lm-aware variant used by RTPS/MVMVA. pub fn satIr16(v: i64) i64 { if (v < -32768) { return -32768; } if (v > 32767) { return 32767; } @@ -694,7 +690,7 @@ pub fn satIrLm(v: i64, lm: u32) i64 { return v; } -// FLAG-aware clamps (psxe gte_clamp_* / gte_check_mac, cpu.c:1486-1607) +// FLAG-aware clamps. // Each ORs the matching FLAG bit into g[32+C_FLAG] on saturation/overflow // and otherwise returns exactly what the plain sat* helpers return, so // MAC/IR/SXY/RGB *values* are unchanged — only FLAG gets populated. `i` @@ -719,28 +715,28 @@ pub fn gteIrTriple(g: *mut[] u32, m1: i64, m2: i64, m3: i64, lm: u32) { g[11] = (gteClampIr(g, 3, m3, lm) as u64 & 0xFFFF) as u32; } -// IR0 clamp 0..0x1000 → bit 12 (gte_clamp_ir0). +// IR0 clamp 0..0x1000 → bit 12. pub fn gteClampIr0(g: *mut[] u32, v: i64) i64 { if (v < 0) { g[32 + C_FLAG] = g[32 + C_FLAG] | 0x1000; return 0; } if (v > 0x1000) { g[32 + C_FLAG] = g[32 + C_FLAG] | 0x1000; return 0x1000; } return v; } -// SZ3 saturation 0..0xFFFF → bit 18 (gte_clamp_sz3). +// SZ3 saturation 0..0xFFFF → bit 18. pub fn gteClampSz3(g: *mut[] u32, v: i64) u32 { if (v < 0) { g[32 + C_FLAG] = g[32 + C_FLAG] | 0x40000; return 0; } if (v > 0xFFFF) { g[32 + C_FLAG] = g[32 + C_FLAG] | 0x40000; return 0xFFFF; } return (v as u64 & 0xFFFF) as u32; } -// SX/SY saturation -0x400..0x3FF → bit 14/13 (gte_clamp_sxy). +// SX/SY saturation -0x400..0x3FF → bit 14/13. pub fn gteClampSxy(g: *mut[] u32, i: u32, v: i64) u32 { if (v < -1024) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x4000 >> (i - 1)); return 0xFFFFFC00; } if (v > 1023) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x4000 >> (i - 1)); return 0x3FF; } return ((v + 0x100000000) as u64 & 0xFFFF) as u32; } -// RGB channel clamp 0..0xFF → bit 21/20/19 (gte_clamp_rgb). +// RGB channel clamp 0..0xFF → bit 21/20/19. pub fn gteClampRgb(g: *mut[] u32, i: u32, v: i64) u32 { if (v < 0) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x200000 >> (i - 1)); return 0; } if (v > 255) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x200000 >> (i - 1)); return 255; } @@ -748,7 +744,7 @@ pub fn gteClampRgb(g: *mut[] u32, i: u32, v: i64) u32 { } // MAC0 overflow → bit 15 (neg) / 16 (pos); value is NOT clamped, only -// flagged (gte_clamp_mac0). +// flagged. pub fn gteClampMac0(g: *mut[] u32, v: i64) i64 { const lim: i64 = (1 as i64) << 31; if (v < (0 - lim)) { g[32 + C_FLAG] = g[32 + C_FLAG] | 0x8000; } @@ -757,23 +753,23 @@ pub fn gteClampMac0(g: *mut[] u32, v: i64) i64 { } // MAC1..MAC3: flag 44-bit overflow (bit 30/29/28 pos, 27/26/25 neg), -// truncate to 44 bits (sign-extend from bit 43), then >> sf. This is -// psxe's gte_clamp_mac — the truncation only changes the result when the -// sum exceeds 44 bits, so normal geometry is unaffected. +// truncate to 44 bits (sign-extend from bit 43), then >> sf. The +// truncation only changes the result when the sum exceeds 44 bits, so +// normal geometry is unaffected. pub fn gteClampMac(g: *mut[] u32, i: u32, v: i64, sf: u32) i64 { const lim: i64 = (1 as i64) << 43; if (v < (0 - lim)) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x8000000 >> (i - 1)); } if (v > (lim - 1)) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x40000000 >> (i - 1)); } const t: i64 = (v << 20) >> 20; - // psxe gte_clamp_mac returns (int32_t)(...): MAC1..3 are 32-bit, so - // truncate before this feeds the IR clamp (a >32-bit result otherwise - // clamps IR / sets the saturation flag from the wrong magnitude). + // MAC1..3 are 32-bit, so truncate to int32 before this feeds the IR + // clamp (a >32-bit result otherwise clamps IR / sets the saturation + // flag from the wrong magnitude). return truncI32(t >> sf); } -// Running-accumulation 44-bit check (gte_check_mac): flag overflow and -// truncate to 44 bits, WITHOUT the >> sf (used between the additive terms -// of a 3-term MAC sum so intermediate overflow wraps like psxe). +// Running-accumulation 44-bit check: flag overflow and truncate to 44 +// bits, WITHOUT the >> sf (used between the additive terms of a 3-term +// MAC sum so intermediate overflow wraps as on hardware). pub fn gteCheckMac(g: *mut[] u32, i: u32, v: i64) i64 { const lim: i64 = (1 as i64) << 43; if (v < (0 - lim)) { g[32 + C_FLAG] = g[32 + C_FLAG] | (0x8000000 >> (i - 1)); } @@ -782,18 +778,18 @@ pub fn gteCheckMac(g: *mut[] u32, i: u32, v: i64) i64 { } // Truncate an i64 to the signed 32-bit value range (low 32 bits, -// sign-extended). Models psxe's `(int32_t)` casts inside gte_clamp_ir_z. +// sign-extended). Models the `(int32_t)` casts inside gteClampIrZ. pub fn truncI32(v: i64) i64 { const lo: u32 = (v as u64 & 0xFFFFFFFF) as u32; return s32ToI64(lo); } -// IR3 clamp for RTPS/RTPT (psxe gte_clamp_ir_z, cpu.c:1595-1607). The -// returned value clamps `value >> sf` to [lm?0:-0x8000, 0x7FFF], but FLAG -// bit 22 is keyed off `value >> 12` (each int32-truncated first). At sf=12 -// the two shifts agree; at sf=0 they diverge — so a plain gteClampIr on the -// shifted MAC sets bit 22 from the wrong magnitude. Note clamp_ir_z does -// NOT flag on the value-clamp itself, only on the >>12 range check. +// IR3 clamp for RTPS/RTPT. The returned value clamps `value >> sf` to +// [lm?0:-0x8000, 0x7FFF], but FLAG bit 22 is keyed off `value >> 12` +// (each int32-truncated first). At sf=12 the two shifts agree; at sf=0 +// they diverge — so a plain gteClampIr on the shifted MAC sets bit 22 +// from the wrong magnitude. Note this clamp does NOT flag on the +// value-clamp itself, only on the >>12 range check. pub fn gteClampIrZ(g: *mut[] u32, value: i64, sf: u32, lm: u32) i64 { const valueSf: i64 = truncI32(value >> sf); const value12: i64 = truncI32(value >> 12); @@ -823,8 +819,8 @@ pub fn satSxy(v: i64) u32 { return lo; } -// Count leading zero bits of a 32-bit value (32 for 0). Matches the -// `clz` helper psxe uses inside gte_divide. +// Count leading zero bits of a 32-bit value (32 for 0). Used by the +// `clz` step inside gteDivide. pub fn gteClz32(v: u32) u32 { if (v == 0) { return 32; } var n: u32 = 0; @@ -837,7 +833,7 @@ pub fn gteClz32(v: u32) u32 { } // GTE perspective divide — the hardware's Unsigned Newton-Raphson -// reciprocal (psxe gte_divide, cpu.c:1616-1634). On overflow (n >= 2d) +// reciprocal. On overflow (n >= 2d) // it returns 0x1FFFF and sets FLAG bit 17 (divide overflow). `n` is H, // `d` is SZ3; both are treated as unsigned 16-bit. pub fn gteDivide(g: *mut[] u32, h: u32, sz3: u32) u32 { @@ -861,9 +857,8 @@ pub fn gteDivide(g: *mut[] u32, h: u32, sz3: u32) u32 { // RTPS — rotate, translate, perspective single. Computes a screen- // space projection of vertex `vidx` and pushes it onto the SXY/SZ -// FIFO. Faithful port of psxe's GTE_RTP_DQ macro including the nested -// 44-bit MAC truncation, the clamp_ir_z IR3 path, and the full FLAG -// bookkeeping (cpu.c:1728-1748). +// FIFO. Includes the nested 44-bit MAC truncation, the gteClampIrZ IR3 +// path, and the full FLAG bookkeeping, matching hardware. pub fn gteRtps(g: *mut[] u32, vidx: u32, sf: u32, lm: u32, last: u32) { const vSlot: u32 = vidx * 2; // V0 at 0, V1 at 2, V2 at 4 const vxy: u32 = g[vSlot]; @@ -872,8 +867,8 @@ pub fn gteRtps(g: *mut[] u32, vidx: u32, sf: u32, lm: u32, last: u32) { const vy: i64 = s16ToI64((vxy >> 16) & 0xFFFF); const vzS: i64 = s16ToI64(vz & 0xFFFF); - // Rotation matrix elements (packed two-per-u32 like psxe's - // gte_matrix_t). Layout: + // Rotation matrix elements (packed two-per-u32 in the control regs). + // Layout: // C[0] low = RT11, high = RT12 // C[1] low = RT13, high = RT21 // C[2] low = RT22, high = RT23 @@ -894,27 +889,27 @@ pub fn gteRtps(g: *mut[] u32, vidx: u32, sf: u32, lm: u32, last: u32) { const try_: i64 = s32ToI64(g[32 + 6]); const trz: i64 = s32ToI64(g[32 + 7]); - // MAC = TR*4096 + R*V, with psxe's per-term 44-bit truncation nested - // between the additive terms (gte_check_mac) so an intermediate that - // crosses 44 bits wraps before the next add — exactly like the GTE_RTP* - // macros (cpu.c:1732-1734). Normal geometry never crosses 44 bits, so - // the values are unchanged there; only the overflow FLAG/wrap differs. + // MAC = TR*4096 + R*V, with per-term 44-bit truncation nested between + // the additive terms (gteCheckMac) so an intermediate that crosses 44 + // bits wraps before the next add, matching hardware. Normal geometry + // never crosses 44 bits, so the values are unchanged there; only the + // overflow FLAG/wrap differs. const c1a: i64 = gteCheckMac(g, 1, (trx << 12) + rt11 * vx); const c1b: i64 = gteCheckMac(g, 1, c1a + rt12 * vy); const sm1: i64 = gteClampMac(g, 1, c1b + rt13 * vzS, sf); const c2a: i64 = gteCheckMac(g, 2, (try_ << 12) + rt21 * vx); const c2b: i64 = gteCheckMac(g, 2, c2a + rt22 * vy); const sm2: i64 = gteClampMac(g, 2, c2b + rt23 * vzS, sf); - // macRaw3 is the raw (pre-shift) MAC3 sum — psxe stores it as s_mac3 - // and feeds it to BOTH gte_clamp_sz3 and gte_clamp_ir_z (cpu.c:1737,1741). + // macRaw3 is the raw (pre-shift) MAC3 sum — it feeds BOTH gteClampSz3 + // and gteClampIrZ. const c3a: i64 = gteCheckMac(g, 3, (trz << 12) + rt31 * vx); const c3b: i64 = gteCheckMac(g, 3, c3a + rt32 * vy); const macRaw3: i64 = c3b + rt33 * vzS; const sm3: i64 = gteClampMac(g, 3, macRaw3, sf); - // IR1/IR2 clamp the shifted MAC; IR3 uses the special clamp_ir_z which - // keys FLAG bit 22 off macRaw3>>12 while clamping macRaw3>>sf - // (cpu.c:1737) — differs from a plain IR clamp only when sf=0. + // IR1/IR2 clamp the shifted MAC; IR3 uses the special gteClampIrZ + // which keys FLAG bit 22 off macRaw3>>12 while clamping macRaw3>>sf + // — differs from a plain IR clamp only when sf=0. const ir1: i64 = gteClampIr(g, 1, sm1, lm); const ir2: i64 = gteClampIr(g, 2, sm2, lm); const ir3: i64 = gteClampIrZ(g, macRaw3, sf, lm); @@ -934,8 +929,8 @@ pub fn gteRtps(g: *mut[] u32, vidx: u32, sf: u32, lm: u32, last: u32) { // SXY = (OFX + IR * div) >> 16. OFX/OFY are 32-bit signed. const ofx: i64 = s32ToI64(g[32 + 24]); const ofy: i64 = s32ToI64(g[32 + 25]); - // psxe wraps OFX+IR*div in gte_clamp_mac0 (MAC0 overflow flag) before - // the >>16, then gte_clamp_sxy (bit 14/13). + // OFX+IR*div is wrapped in gteClampMac0 (MAC0 overflow flag) before + // the >>16, then gteClampSxy (bit 14/13). const sx: i64 = gteClampMac0(g, ofx + ir1 * (div as i64)) >> 16; const sy: i64 = gteClampMac0(g, ofy + ir2 * (div as i64)) >> 16; @@ -944,10 +939,10 @@ pub fn gteRtps(g: *mut[] u32, vidx: u32, sf: u32, lm: u32, last: u32) { g[D_SXY1] = g[D_SXY2]; g[D_SXY2] = (gteClampSxy(g, 1, sx) & 0xFFFF) | ((gteClampSxy(g, 2, sy) & 0xFFFF) << 16); - // Write IR back and MAC for further pipeline stages. psxe's - // gte_clamp_mac stores the *shifted* (>> sf) value into MAC1..MAC3 - // (cpu.c:1508), so we store sm1..sm3 (= mac >> sf), NOT the raw - // pre-shift sum. SZ3 above still uses the unshifted mac3 >> 12. + // Write IR back and MAC for further pipeline stages. gteClampMac + // stores the *shifted* (>> sf) value into MAC1..MAC3, so we store + // sm1..sm3 (= mac >> sf), NOT the raw pre-shift sum. SZ3 above still + // uses the unshifted mac3 >> 12. g[9] = (ir1 as u64 & 0xFFFF) as u32; g[10] = (ir2 as u64 & 0xFFFF) as u32; g[11] = (ir3 as u64 & 0xFFFF) as u32; @@ -955,25 +950,25 @@ pub fn gteRtps(g: *mut[] u32, vidx: u32, sf: u32, lm: u32, last: u32) { g[D_MAC2] = (sm2 as u64 & 0xFFFFFFFF) as u32; g[D_MAC3] = (sm3 as u64 & 0xFFFFFFFF) as u32; - // Depth-queue tail (psxe's GTE_RTP_DQ). Only the LAST vertex runs it: - // psxe's RTPT does GTE_RTP(0); GTE_RTP(1); GTE_RTP_DQ(2) (cpu.c:2247-2251) - // and duckstation gates the DQ block on `last` (gte.cpp:885). Running it on - // vertices 0/1 would set spurious FLAG bits 12/15/16 (15/16 feed bit 31), - // so RTPT could report a false error. DQA is s16 (cpu.c:138 / - // gte_types.h:117) — sign-extend the low half, not the full 32 bits. + // Depth-queue tail. Only the LAST vertex runs it: RTPT does RTP on + // vertices 0 and 1, then RTP+DQ on vertex 2, and DuckStation gates the + // DQ block on `last` (gte.cpp:885). Running it on vertices 0/1 would + // set spurious FLAG bits 12/15/16 (15/16 feed bit 31), so RTPT could + // report a false error. DQA is s16 — sign-extend the low half, not the + // full 32 bits. if (last != 0) { const dqa: i64 = s16ToI64(g[32 + 27]); const dqb: i64 = s32ToI64(g[32 + 28]); const mac0: i64 = gteClampMac0(g, dqb + dqa * (div as i64)); g[D_MAC0] = (mac0 as u64 & 0xFFFFFFFF) as u32; - // IR0 clamps to 0..0x1000 with FLAG bit 12 (gte_clamp_ir0). + // IR0 clamps to 0..0x1000 with FLAG bit 12. const ir0: i64 = gteClampIr0(g, mac0 >> 12); g[8] = (ir0 as u64 & 0xFFFF) as u32; } } // AVSZ3 / AVSZ4 — average Z of the last 3 or 4 SZ FIFO entries, -// scaled by ZSF3 / ZSF4 respectively. psxe: +// scaled by ZSF3 / ZSF4 respectively: // AVSZ3: MAC0 = ZSF3 * (SZ1 + SZ2 + SZ3); OTZ = clampSZ(MAC0 >> 12) // AVSZ4: MAC0 = ZSF4 * (SZ0 + SZ1 + SZ2 + SZ3); OTZ = clampSZ(MAC0 >> 12) // ZSF3 is at control reg 29 (s16), ZSF4 at control reg 30 (s16). @@ -991,8 +986,7 @@ pub fn gteAvsz(g: *mut[] u32, count: u32) { // NCLIP — back-face cull test. MAC0 = signed cross product of the three // screen-space vertices currently sitting in the SXY FIFO. The BIOS -// uses the sign of MAC0 to decide whether to draw a triangle. Port of -// psx_gte_i_nclip in psxe. +// uses the sign of MAC0 to decide whether to draw a triangle. pub fn gteNclip(g: *mut[] u32) { const sx0: i64 = s16ToI64(g[D_SXY0] & 0xFFFF); const sy0: i64 = s16ToI64((g[D_SXY0] >> 16) & 0xFFFF); @@ -1004,7 +998,7 @@ pub fn gteNclip(g: *mut[] u32) { g[D_MAC0] = (v as u64 & 0xFFFFFFFF) as u32; } -// OP — outer product / cross product variant. psxe: +// OP — outer product / cross product variant: // MAC1 = RT22*IR3 - RT33*IR2 // MAC2 = RT33*IR1 - RT11*IR3 // MAC3 = RT11*IR2 - RT22*IR1 @@ -1017,7 +1011,7 @@ pub fn gteOp(g: *mut[] u32, sf: u32, lm: u32) { const ir1: i64 = s16ToI64(g[9]); const ir2: i64 = s16ToI64(g[10]); const ir3: i64 = s16ToI64(g[11]); - // psxe runs each through gte_clamp_mac (>>sf + flag) then gte_clamp_ir. + // Each runs through gteClampMac (>>sf + flag) then the IR clamp. const mac1: i64 = gteClampMac(g, 1, rt22 * ir3 - rt33 * ir2, sf); const mac2: i64 = gteClampMac(g, 2, rt33 * ir1 - rt11 * ir3, sf); const mac3: i64 = gteClampMac(g, 3, rt11 * ir2 - rt22 * ir1, sf); @@ -1029,7 +1023,7 @@ pub fn gteOp(g: *mut[] u32, sf: u32, lm: u32) { // NCDS — Normal Color Depth Single. Computes per-vertex lit color with // far-color (fog) depth interpolation, using the light matrix to -// transform the input vertex normal. Mirrors psxe's NCDS macro: +// transform the input vertex normal. The pipeline: // 1. L * V → MAC, IR (vertex normal in light space) // 2. BK + LC*IR → MAC, IR (background + reflected light) // 3. ir' = clamp((FC<<12) - (C<<4)*IR), with lm=0 @@ -1050,9 +1044,9 @@ pub fn gteOp(g: *mut[] u32, sf: u32, lm: u32) { // // Without this, the BIOS PS-logo polygons all receive colour (0,0,0) // and render black on a black background — completely invisible. -// sf is 0 or 12 — psxe's gte_clamp_mac shifts the saturated MAC by sf -// at every stage; we apply the same shift inline so IR values land in -// the right magnitude before clamping. +// sf is 0 or 12 — gteClampMac shifts the saturated MAC by sf at every +// stage; we apply the same shift inline so IR values land in the right +// magnitude before clamping. pub fn gteNcds(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { const vSlot: u32 = vidx * 2; const vxy: u32 = g[vSlot]; @@ -1072,10 +1066,10 @@ pub fn gteNcds(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { const ll32: i64 = s16ToI64((g[32 + 11] >> 16) & 0xFFFF); const ll33: i64 = s16ToI64(g[32 + 12] & 0xFFFF); - // Stage 1: L * V → MAC, IR. psxe's gte_clamp_mac applies `>> sf`, - // so for sf=12 (the BIOS default) we shift down by 12 before - // clamping into the IR range; otherwise IR saturates at 0x7FFF and - // the downstream stages clip every channel to white. + // Stage 1: L * V → MAC, IR. gteClampMac applies `>> sf`, so for + // sf=12 (the BIOS default) we shift down by 12 before clamping into + // the IR range; otherwise IR saturates at 0x7FFF and the downstream + // stages clip every channel to white. var m1a: i64 = ll11 * vx + ll12 * vy + ll13 * vz; var m2a: i64 = ll21 * vx + ll22 * vy + ll23 * vz; var m3a: i64 = ll31 * vx + ll32 * vy + ll33 * vz; @@ -1098,8 +1092,8 @@ pub fn gteNcds(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { const gbk: i64 = s32ToI64(g[32 + 14]); const bbk: i64 = s32ToI64(g[32 + 15]); - // BK<<12 + LC*IR with the same nested 44-bit truncation psxe applies - // (NCDS macro, cpu.c:1855-1857). + // BK<<12 + LC*IR with the same nested 44-bit truncation as in the + // L*V stage. const b1a: i64 = gteCheckMac(g, 1, (rbk << 12) + lr1 * ir1a); const b1b: i64 = gteCheckMac(g, 1, b1a + lr2 * ir2a); const m1b: i64 = gteClampMac(g, 1, b1b + lr3 * ir3a, sf); @@ -1120,7 +1114,7 @@ pub fn gteNcds(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { const bc: i64 = ((rgbc >> 16) & 0xFF) as i64; // Stage 4: far-colour interpolation ir' = clamp((FC<<12) - (C<<4)*IR). - // psxe shifts this stage by `sf` too via the same gte_clamp_mac. + // This stage is shifted by `sf` too, via the same gteClampMac. const rfc: i64 = s32ToI64(g[32 + 21]); const gfc: i64 = s32ToI64(g[32 + 22]); const bfc: i64 = s32ToI64(g[32 + 23]); @@ -1144,14 +1138,14 @@ pub fn gteNcds(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { g[D_MAC3] = (m3c as u64 & 0xFFFFFFFF) as u32; gteIrTriple(g, m1c, m2c, m3c, lm); - // Stage 6: RGB FIFO push. Use gtePushRgb (FLAG-aware gte_clamp_rgb) so the - // colour-saturation FLAG bits 19/20/21 are set like psxe (cpu.c:1917-1919) - // and duckstation (gte.cpp:209-224); the old inline clampU8 skipped them. + // Stage 6: RGB FIFO push. Use gtePushRgb (FLAG-aware gteClampRgb) so the + // colour-saturation FLAG bits 19/20/21 are set as on hardware and in + // DuckStation (gte.cpp:209-224); the old inline clampU8 skipped them. gtePushRgb(g, m1c, m2c, m3c); } // SQR — square each IR component. Used for vector-length computations. -// psxe: MAC{1,2,3} = IR{1,2,3}^2; IR{1,2,3} = clamp(MAC). +// MAC{1,2,3} = IR{1,2,3}^2; IR{1,2,3} = clamp(MAC). pub fn gteSqr(g: *mut[] u32, sf: u32, lm: u32) { const ir1: i64 = s16ToI64(g[9]); const ir2: i64 = s16ToI64(g[10]); @@ -1171,8 +1165,8 @@ pub fn gteSqr(g: *mut[] u32, sf: u32, lm: u32) { // (cv: 0=TR, 1=BK, 2=FC, 3=zero). Bit 19 sets sf (shift fraction = 12 // or 0). Bit 10 sets lm (limit IR to 0..0x7FFF if 1). // -// This omits the buggy CV=FC case from psxe — most BIOS calls use -// CV=TR. Result: MAC = (CV << 12) + MX * V; IR = clamp(MAC >> sf). +// The CV=FC case is the hardware bug case — most BIOS calls use CV=TR. +// Result: MAC = (CV << 12) + MX * V; IR = clamp(MAC >> sf). pub fn gteMvmva(g: *mut[] u32, opc: u32) { const sf: u32 = ((opc >> 19) & 1) * 12; const lm: u32 = (opc >> 10) & 1; @@ -1181,9 +1175,9 @@ pub fn gteMvmva(g: *mut[] u32, opc: u32) { const mxSel: u32 = (opc >> 17) & 3; // Pick the matrix. mxSel 0/1/2 = rotation / light / light-colour at - // control regs 0..4 / 8..12 / 16..20. mxSel 3 is psxe's "garbage - // matrix" (cpu.c:1988-1998): a junk mix of RC/IR0/RT13/RT22 that the - // hardware actually produces — port it exactly rather than approximate. + // control regs 0..4 / 8..12 / 16..20. mxSel 3 is the hardware's + // "garbage matrix": a junk mix of RC/IR0/RT13/RT22 that the hardware + // actually produces — reproduce it exactly rather than approximate. var m11: i64 = 0; var m12: i64 = 0; var m13: i64 = 0; var m21: i64 = 0; var m22: i64 = 0; var m23: i64 = 0; var m31: i64 = 0; var m32: i64 = 0; var m33: i64 = 0; @@ -1248,11 +1242,10 @@ pub fn gteMvmva(g: *mut[] u32, opc: u32) { } // cvSel == 3: zero translation. - // MAC = (CV<<12) + M*V with psxe's nested 44-bit truncation between the - // additive terms (gte_check_mac, cpu.c:2038-2040). CV=FC (cvSel==2) is - // the hardware bug case: the (CV<<12 + M*VX) column is computed only for - // its FLAG side effects and dropped from the result, leaving the VY/VZ - // columns (cpu.c:2024-2036). + // MAC = (CV<<12) + M*V with nested 44-bit truncation between the + // additive terms (gteCheckMac). CV=FC (cvSel==2) is the hardware bug + // case: the (CV<<12 + M*VX) column is computed only for its FLAG side + // effects and dropped from the result, leaving the VY/VZ columns. var sm1: i64 = 0; var sm2: i64 = 0; var sm3: i64 = 0; if (cvSel == 2) { const k1: i64 = gteCheckMac(g, 1, m12 * vy); @@ -1261,8 +1254,8 @@ pub fn gteMvmva(g: *mut[] u32, opc: u32) { sm2 = gteClampMac(g, 2, k2 + m23 * vz, sf); const k3: i64 = gteCheckMac(g, 3, m32 * vy); sm3 = gteClampMac(g, 3, k3 + m33 * vz, sf); - // Dropped VX column: clamp_mac + clamp_ir purely to raise FLAG bits - // (the results are discarded, exactly as psxe does at cpu.c:2030-2036). + // Dropped VX column: gteClampMac + gteClampIr purely to raise FLAG + // bits (the results are discarded, matching hardware). const d1: i64 = gteClampMac(g, 1, (tx << 12) + m11 * vx, sf); const d2: i64 = gteClampMac(g, 2, (ty << 12) + m21 * vx, sf); const d3: i64 = gteClampMac(g, 3, (tz << 12) + m31 * vx, sf); diff --git a/irq.jam b/irq.jam index 9a2f20f..1b1dc0c 100644 --- a/irq.jam +++ b/irq.jam @@ -1,4 +1,4 @@ -// Interrupt controller (port of psxe/psx/dev/ic.c). +// Interrupt controller. // // Two registers live at 0x1F801070..0x1F801077: // 0x00 I_STAT status — set by devices raising IRQs, cleared by @@ -14,7 +14,7 @@ const { Vec } = import("std/collections"); -// IRQ line bits (psxe ic.h: IC_VBLANK = 0x001, etc.). +// IRQ line bits (IC_VBLANK = 0x001, etc.). pub const IC_VBLANK: u32 = 0x001; pub const IC_GPU: u32 = 0x002; pub const IC_CDROM: u32 = 0x004; diff --git a/main.jam b/main.jam index 97fa995..0ebc048 100644 --- a/main.jam +++ b/main.jam @@ -124,8 +124,7 @@ const SCREEN_W: i32 = 640; const SCREEN_H: i32 = 480; // Decode the GPU's display-mode register (set by GP1 0x08) into -// horizontal and vertical pixel counts. Mirrors psxe's -// `psx_get_dmode_width` / `psx_get_dmode_height` so the SDL window +// horizontal and vertical pixel counts so the SDL window // shows whatever resolution the BIOS or game has currently selected. // // Bit layout of GP1 0x08: @@ -145,12 +144,12 @@ fn dmodeWidth(dm: u32) i32 { } fn dmodeHeight(dm: u32, gpuState: *mut[] u32) i32 { - // Vertical 480 requires interlace (bit 5) + bit 2 set per psxe. + // Vertical 480 requires interlace (bit 5) + bit 2 set. if ((dm & 0x04) != 0 && (dm & 0x20) != 0) { return 480; } - // psxe: when bit 2 alone is set, also 480. + // When bit 2 alone is set, also 480. if ((dm & 0x04) != 0) { return 480; } // Otherwise compute from GP1(07) vertical range. g[67]=disp_y1, - // g[68]=disp_y2. Mirrors psxe psx_get_dmode_height. + // g[68]=disp_y2. const y1: i32 = gpuState[67] as i32; const y2: i32 = gpuState[68] as i32; const disp: i32 = y2 - y1; @@ -236,20 +235,19 @@ fn loadBios(path: []u8, bus: Bus) u64 { return 0; } -// Run one full NTSC frame, mirroring psxe's GPU timing EXACTLY. +// Run one full NTSC frame with cycle-accurate GPU timing. // -// psxe runs the GPU on its own 53.693175 MHz clock (gpu.c psx_gpu_update): -// a scanline is 3413 GPU cycles and HBlank spans [2560, 3413] (the last -// ~25%). Each instruction advances a *float* accumulator by +// The GPU runs on its own 53.693175 MHz clock: a scanline is 3413 GPU +// cycles and HBlank spans [2560, 3413] (the last ~25%). Each instruction +// advances a *float* accumulator by // `delta_cpu_cycles × (53.693175 / 33.868800)` GPU cycles. The HBlank // event — line++, GPUSTAT bit-31 (odd/even), VBlank IRQ — fires on the // crossing INTO [2560,3413]; the scanline wraps (acc -= 3413) on the -// crossing back out. We replicate this in f32 so the sub-cycle phase -// tracks psxe bit-for-bit: an integer cycles/scanline drifts ~0.03 +// crossing back out. The f32 accumulator preserves the sub-cycle phase: +// an integer cycles/scanline drifts ~0.03 // cyc/line and by ~19M instructions mistimes a VBlank IRQ enough to -// diverge (found via REG_TRACE cmp — see jamstation-psxe-parity memory). -// jam's f32 behaviour on these exact values is pinned in -// ../jam/tests/unit/test_float_psxe_parity.jam. +// diverge (found via REG_TRACE cmp). +// jam's f32 behaviour on these exact values is pinned by a unit test. const GPU_CYCLES_HDRAW: f32 = 2560.0; const GPU_CYCLES_SCANL: f32 = 3413.0; const SCANLINES_VDRAW: u32 = 240; @@ -266,14 +264,14 @@ const FloatBits = union { i: u32, f: f32 }; fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, audioDev: u32, ring: *mut[] u8) u32 { // The scanline counter and the float GPU-cycle accumulator are - // CONTINUOUS across frames, exactly like psxe's free-running gpu->line - // / gpu->cycles. Persist them in GPU scratch slots 36/37 (the f32 via + // CONTINUOUS across frames — they free-run rather than resetting per + // frame. Persist them in GPU scratch slots 36/37 (the f32 via // a u32 type-pun) so the sub-cycle phase doesn't reset every frame. var gpuBuf: *mut[] u32 = bus.gpu.bufPtr(); var line: u32 = gpuBuf[36]; var accB: FloatBits = FloatBits { i: gpuBuf[37] }; var acc: f32 = accB.f; - // psxe gpu.c: GPU cycles per CPU cycle = 53.693175 / 33.868800 MHz. + // GPU cycles per CPU cycle = 53.693175 / 33.868800 MHz. const ratio: f32 = (53.693175 as f32) / (33.868800 as f32); // SPU sample accumulator: 1 sample per 768 CPU cycles (33.8688MHz/44100). // Drives the ADSR envelope CPU-synchronously (slot 35, persisted). @@ -290,25 +288,24 @@ fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, const before: u64 = c.cycles; step(c, bus, regs, cop0); const delta: u32 = (c.cycles - before) as u32; - // Per-instruction device order MIRRORS psxe psx_update (psx.c:95-99) - // EXACTLY: cdrom → gpu → pad → timer → dma. (Was timer→dma→cdrom→…→gpu; - // the order decides which IRQ bit lands in I_STAT first each step, and - // psxe fires the GPU hblank/vblank events BEFORE the sysclk timer +2.) + // Per-instruction device order: cdrom → gpu → pad → timer → dma. + // (Was timer→dma→cdrom→…→gpu; the order decides which IRQ bit lands + // in I_STAT first each step, and the GPU hblank/vblank events must + // fire BEFORE the sysclk timer +2.) - // 1. CDROM (psx.c:95) + // 1. CDROM bus.cdrom.update(bus.disc.ptrMut(), bus.spu.ptr, bus.irq.ptr, cop0, delta); - // 2. GPU (psx.c:96) — psx_gpu_update: scanline window [0,3413], HBlank - // [2560,3413]. The HBlank event (line++, GPUSTAT bit-31 odd/even, the - // hblank/vblank timer hooks + VBlank IRQ) fires on the crossing INTO - // [2560,3413] (~75% through the line); the scanline wraps on crossing - // out. This runs BEFORE the sysclk timer update, matching psxe. + // 2. GPU — scanline window [0,3413], HBlank [2560,3413]. The HBlank + // event (line++, GPUSTAT bit-31 odd/even, the hblank/vblank timer + // hooks + VBlank IRQ) fires on the crossing INTO [2560,3413] (~75% + // through the line); the scanline wraps on crossing out. This runs + // BEFORE the sysclk timer update. const prevHb: bool = acc >= GPU_CYCLES_HDRAW && acc <= GPU_CYCLES_SCANL; acc = acc + (delta as f32) * ratio; const currHb: bool = acc >= GPU_CYCLES_HDRAW && acc <= GPU_CYCLES_SCANL; if (currHb && !prevHb) { - // psxe gpu_hblank_event (gpu.c:1908-1962), fired from the - // HBlank-begin edge. + // GPU HBlank event, fired from the HBlank-begin edge. bus.timer.hblankBegin(bus.irq.ptr, cop0); if (line < SCANLINES_VDRAW) { if ((line & 1) != 0) { @@ -329,24 +326,24 @@ fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, frameDone = true; } } else if (prevHb && !currHb) { - // HBlank-end edge → wrap the scanline accumulator (psxe: - // gpu->cycles -= 3413). + // HBlank-end edge → wrap the scanline accumulator + // (acc -= 3413). bus.timer.hblankEnd(); acc = acc - GPU_CYCLES_SCANL; } - // 3. PAD (psx.c:97) + // 3. PAD padUpdate(bus.pad.ptr, bus.irq.ptr, cop0, delta); - // 4. TIMER sysclk (psx.c:98) — updateCyc ignores delta, uses 2 (psxe). + // 4. TIMER sysclk — updateCyc ignores delta, uses 2. bus.timer.updateCyc(bus.irq.ptr, cop0, delta); - // 5. DMA (psx.c:99) + // 5. DMA dmaTickSpu(bus.dma.ptr, delta); dmaUpdate(bus.dma.ptr, bus.irq.ptr, cop0); - // jam-only extras: psxe steps NEITHER in psx_update. Kept at the - // end (least perturbation to the psxe device order above) and flagged - // as divergences — SIO1 → peripherals task, SPU CPU-clock → SPU task - // (psxe runs SPU on the audio thread; revisit once the deterministic - // CPU surface matches psxe). + // jam-only extras: neither of these belongs in the per-instruction + // device update above. Kept at the end (least perturbation to the + // device order above) and flagged as divergences — SIO1 → + // peripherals task, SPU CPU-clock → SPU task (revisit once the + // deterministic CPU surface settles). bus.sio1.update(bus.irq.ptr, cop0, delta); spuAcc = spuAcc + delta; while (spuAcc >= 768) { @@ -460,14 +457,13 @@ fn main() { // Memory card slot 1 — pull saved data from ./slot1.mcd if it // exists. On exit we write the (possibly modified) 128 KB buffer - // back to the same file so saves survive across runs. Mirrors - // psxe's psx_mcd_init / psx_mcd_destroy lifecycle. + // back to the same file so saves survive across runs. var mcdPath: []u8 = "./slot1.mcd"; if (mcdRamLoad(bus.mcdRam.ptr, mcdPath)) { print("Loaded memcard: ./slot1.mcd\n"); } - // Match psxe's initial COP0 state. The 0x10900000 value seeds COP0 + // Initial COP0 state. The 0x10900000 value seeds COP0 // SR with CU0 enabled, BEV set, and a few breakpoint controls. cop0Write(cop0.ptr, C0_SR, 0x10900000); cop0Write(cop0.ptr, C0_PRID, 0x00000002); @@ -510,7 +506,7 @@ fn main() { // that depend on the BIOS kernel being set up, drop the EXE at // `./game.exe` instead: we boot the BIOS first and only sideload // when PC reaches 0x80030000 (the standard "kernel ready" address - // psxe uses for the same handoff). + // for this handoff). var demoPath = "./demo.exe"; var gamePath = "./game.exe"; var cuePath: []u8 = "./game.cue"; @@ -721,8 +717,7 @@ fn main() { } } - // Persist memcard slot 1 to disk before freeing the bus. psxe does - // the equivalent in psx_mcd_destroy. + // Persist memcard slot 1 to disk before freeing the bus. mcdRamSave(bus.mcdRam.ptr, mcdPath); sdlAudioClose(audioDev); diff --git a/mcd.jam b/mcd.jam index 5faaaae..c4d4784 100644 --- a/mcd.jam +++ b/mcd.jam @@ -1,4 +1,4 @@ -// Memory Card (PSX BU01) — port of psxe/psx/dev/mcd.c. +// Memory Card (PSX BU01). // // Memory cards live on SIO0 as a second device alongside the joypad. // When the host writes 0x81 to SIO0_DATA the bus selects the memcard; @@ -6,7 +6,7 @@ // 128 bytes per frame. Our pad.jam handles slot selection — it routes // 0x81 dest writes through here. // -// State machine (psxe mcd.c MCD_STATE_*): +// State machine (MCD_STATE_*): // // TX_HIZ → 0xFF on first read (line high-Z) // TX_FLG → flag byte (0x08 = "first access since power-on") @@ -27,7 +27,7 @@ const { File } = import("std/fs"); const MCD_MEMORY_SIZE: u64 = 0x20000; // 128 KB -// MCD state-machine states (mirrors psxe enum). +// MCD state-machine states. const MCD_TX_HIZ: u32 = 0; const MCD_TX_FLG: u32 = 1; const MCD_TX_ID1: u32 = 2; @@ -104,7 +104,7 @@ pub const Mcd = struct { return self.txReady != 0; } - // Read one byte from the card (psxe psx_mcd_read). The state + // Read one byte from the card. The state // advances after each byte; the host clocks 8 bits per call. // Returns the byte the host should latch into SIO0's RX register. pub fn read(self: mut Self, ram: *mut[] u8) u32 { @@ -112,7 +112,7 @@ pub const Mcd = struct { match (st) { MCD_TX_HIZ { self.txData = 0xFF; } MCD_TX_FLG { - // psxe patch: if the host's last RX was 0x81/0x01 the card + // If the host's last RX was 0x81/0x01 the card // bails out — game-side probe code expects 0xFF when it // tries to ping us by sending the dest byte twice. const rx: u32 = self.rxData as u32; diff --git a/mdec.jam b/mdec.jam index dc77536..c5f24b3 100644 --- a/mdec.jam +++ b/mdec.jam @@ -1,4 +1,4 @@ -// MDEC — Motion Decoder. Port of psxe/psx/dev/mdec.c (~470 lines). +// MDEC — Motion Decoder. // // PSX intros (Musashi SQUARESOFT logo, Harvest Moon opening, BIOS PS-X // startup splash, etc.) decompress YUV macroblock data through this @@ -34,7 +34,7 @@ const DEPTH_8BIT: u32 = 1; const DEPTH_24BIT: u32 = 2; const DEPTH_15BIT: u32 = 3; -// psxe's `int zagzig[]` — index into yblk in iDCT-input order (mdec.c:19-28). +// Zig-zag table — index into yblk in iDCT-input order. const ZAGZIG_INIT: [64]u8 = [ 0, 1, 8, 16, 9, 2, 3, 10, 17, 24, 32, 25, 18, 11, 4, 5, @@ -52,7 +52,7 @@ const ZAGZIG_INIT: [64]u8 = [ // // yBlk / crBlk / cbBlk are 128-element i16 arrays — the iDCT // ping-pongs between the first 64 entries (live block) and the next -// 64 (scratch). That matches psxe's "blk + 128" scratch offset. +// 64 (scratch), using a "blk + 128" scratch offset. pub const Mdec = struct { cmd: u32, wordsRem: u32, @@ -129,7 +129,7 @@ pub const Mdec = struct { }; } - // iDCT (mdec.c:45 real_idct) + // iDCT // // Two-pass 8×8 transform. The block (blk) provides both the input // and final output; we ping-pong through the scratch slot at @@ -137,12 +137,12 @@ pub const Mdec = struct { pub fn realIdct(self: mut Self, blk: *mut[] i16) { idctPassPtr(blk, blk, 64, self.scale.asMutPtr()); idctPassPtr(blk, blk, 64, self.scale.asMutPtr()); - // Note: real_idct ping-pongs blk↔scratch via 64-entry offset. + // Note: the iDCT ping-pongs blk↔scratch via 64-entry offset. // Our 128-entry block holds both. After the two passes the // final result lives in blk[0..64]. } - // RLE block decode (mdec.c:72 rl_decode_block) + // RLE block decode // // Reads 16-bit words from `input` starting at `inIdx`. Decodes one // block into `blk` (a 128-entry i16 array; lower 64 are the live @@ -195,7 +195,7 @@ pub const Mdec = struct { return idx; } - // YUV → RGB (mdec.c:135 yuv_to_rgb) + // YUV → RGB pub fn yuvToRgb(self: mut Self, output: *mut[] u8, outBase: u32, xx: u32, yy: u32) { const depth: u32 = self.outputDepth; @@ -210,8 +210,8 @@ pub const Mdec = struct { const rRaw: i32 = self.crBlk[cri] as i32; const bRaw: i32 = self.cbBlk[cri] as i32; - // psxe uses floats: g = -0.3437*b + -0.7143*r; - // r = 1.402*r; b = 1.772*b. + // YUV→RGB coefficients (float): g = -0.3437*b + -0.7143*r; + // r = 1.402*r; b = 1.772*b. const rf: f64 = rRaw as f64; const bf: f64 = bRaw as f64; const gFloat: f64 = (0.0 - 0.3437) * bf + (0.0 - 0.7143) * rf; @@ -253,7 +253,7 @@ pub const Mdec = struct { } } - // decode macroblock (mdec.c:180) + // decode macroblock pub fn decodeMacroblock(self: mut Self, input: *mut[] u8, output: *mut[] u8) { const depth: u32 = self.outputDepth; @@ -362,7 +362,7 @@ pub const Mdec = struct { } } - // bus dispatch (mdec.c:287/348) + // bus dispatch pub fn read32(self: mut Self, output: *mut[] u8, off: u32) u32 { if (off == 0) { // Data port — drain one word from the output buffer. @@ -380,11 +380,11 @@ pub const Mdec = struct { return 0xAAAAAAAA; } if (off == 4) { - // Status register — psxe mdec.c:316. + // Status register. // Bits 0-15 = "number of parameter words remaining MINUS 1" // (nocash: FFFFh = none). DuckStation `(remaining/2)-1`, Avocado // `(paramCount-1)&0xffff`, mednafen's 0xFFFF sentinel all agree; - // psxe (and the old jam code) reported the raw count, so a game + // the old jam code reported the raw count, so a game // that polls for (status & 0xFFFF)==0xFFFF after a SET_QT/SET_ST // upload to confirm "no params pending" never saw it and never // advanced to DECODE. wordsRem is u32, so 0-1 wraps to 0xFFFF here. @@ -397,10 +397,11 @@ pub const Mdec = struct { // recomputed here, NOT latched: bit 27 = DMA1 enabled AND output // has data; bit 28 = DMA0 enabled AND more input wanted; bit 31 = // output drained. Matches nocash psx-spx, DuckStation - // (`enable_dma_out && !out_fifo.empty()`) and Avocado. psxe latches - // outputReq at the final decode word, so a game that enables DMA1 - // AFTER the decode reads bit 27 = 0 forever and never DMAs the MDEC - // output back out -- the frame decodes but is never read (FMV skip). + // (`enable_dma_out && !out_fifo.empty()`) and Avocado. Latching + // outputReq at the final decode word instead would mean a game that + // enables DMA1 AFTER the decode reads bit 27 = 0 forever and never + // DMAs the MDEC output back out -- the frame decodes but is never + // read (FMV skip). if (self.enableDma1 != 0 && self.outputWords > 0) { st = st | (1 << 27); } if (self.enableDma0 != 0 && self.wordsRem > 0) { st = st | (1 << 28); } if (self.busy != 0) { st = st | (1 << 29); } @@ -541,10 +542,10 @@ pub fn clampS10(v: i32) i32 { return v; } -// Wrap an i32 to int16. psxe rl_decode_block holds the dequant product -// in an int16_t, so it TRUNCATES before the CLAMP(-0x400,0x3ff) -// (mdec.c:90,96,114). For products past int16 range the wrap can flip -// the sign and clamp to the opposite bound — must be applied pre-clamp. +// Wrap an i32 to int16. The RLE dequant product is held in an int16_t, +// so it TRUNCATES before the CLAMP(-0x400,0x3ff). For products past +// int16 range the wrap can flip the sign and clamp to the opposite +// bound — must be applied pre-clamp. pub fn mdecTrunc16(v: i32) i32 { return ((v & 0xFFFF) ^ 0x8000) - 0x8000; } @@ -579,8 +580,8 @@ pub fn writeU32(buf: *mut[] u8, off: u32, v: u32) { // secondary pass. We pingpong by indexing through that offset. pub fn idctPassPtr(blk: *mut[] i16, _dst: *mut[] i16, _scratchOff: u32, scale: *mut[] i16) { - // Pass: write into the scratch half, then copy back. Mirrors - // psxe's two-pass iDCT via a 64-entry temp. + // Pass: write into the scratch half, then copy back. This is the + // two-pass iDCT via a 64-entry temp. var scratch: [64]i16 = [0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, diff --git a/pad.jam b/pad.jam index 5bd8eb0..91d631c 100644 --- a/pad.jam +++ b/pad.jam @@ -1,5 +1,5 @@ -// SIO0 joypad controller (port of psxe/psx/dev/pad.c + the SDA digital -// pad protocol). Register file at 0x1F801040..0x1F80104F: +// SIO0 joypad controller (SDA digital pad protocol). Register file at +// 0x1F801040..0x1F80104F: // 0x40 JOY_TX_FIFO (write) / JOY_RX_FIFO (read) // 0x44 JOY_STAT (read) // 0x48 JOY_MODE @@ -155,8 +155,8 @@ pub fn sdaWrite(p: *mut[] u8, data: u32) { } } -// pad_read_rx: returns the next byte the controller has staged in -// response to the last TX. Mirrors psxe — when CTRL.JOUT and RXEN are +// padReadRx: returns the next byte the controller has staged in +// response to the last TX. When CTRL.JOUT and RXEN are // both clear, the line is high-Z and reads return 0xFFFFFFFF. pub fn padReadRx(p: *mut[] u8, mcd: mut Mcd, mcdRam: *mut[] u8) u32 { const ctrl: u32 = padGetU16(p, P_CTRL_LO); @@ -187,7 +187,7 @@ pub fn padWriteTx(p: *mut[] u8, mcd: mut Mcd, data: u32) { if ((ctrl & CTRL_TXEN) == 0) { return; } if (p[P_DEST] == 0) { // First TX byte selects the destination (0x01 = joy, 0x81 = mcd). - // psxe pad.c:66-78: sets dest, schedules the ACK-pulse countdown + // Sets dest, schedules the ACK-pulse countdown // if CTRL_ACIE — but does NOT set irq_bit on the first byte. The // BIOS expects IC_JOY to assert while STAT.IRQ7 stays 0 until a // *subsequent* byte writes through. @@ -209,12 +209,12 @@ pub fn padWriteTx(p: *mut[] u8, mcd: mut Mcd, data: u32) { } if (p[P_DEST] == DEST_JOY as u8) { sdaWrite(p, data); - // psxe pad.c:90-91: if the controller has nothing more to send, + // If the controller has nothing more to send, // drop the slot so the next first byte re-selects. if (p[P_SDA_TXR] == 0) { p[P_DEST] = 0; } } else if (p[P_DEST] == DEST_MCD as u8) { mcd.write(data); - // psxe pad.c:103-108: MCD always sets irq_bit and uses a 1024- + // MCD always sets irq_bit and uses a 1024- // cycle delay (vs JOY's 512). The longer ACK reflects the MCD's // slower response time. if ((ctrl & CTRL_ACIE) != 0) { @@ -233,7 +233,7 @@ pub fn padWriteTx(p: *mut[] u8, mcd: mut Mcd, data: u32) { } } -// pad_handle_ctrl_write: write to JOY_CTRL. Per psxe: when JOUT is +// padHandleCtrlWrite: write to JOY_CTRL. When JOUT is // cleared, also clear the slot bit; bit 4 (ACKN) is a one-shot that // clears STAT bits 3, 4, 5, 9 and self-clears. pub fn padHandleCtrlWrite(p: *mut[] u8, mcd: mut Mcd, value: u32) { @@ -242,7 +242,7 @@ pub fn padHandleCtrlWrite(p: *mut[] u8, mcd: mut Mcd, value: u32) { var ctrl: u32 = padGetU16(p, P_CTRL_LO); ctrl = ctrl & (~CTRL_SLOT); padSetU16(p, P_CTRL_LO, ctrl); - // psxe pad.c:134-135: !JOUT also resets the slot's MCD state so + // !JOUT also resets the slot's MCD state so // the next 0x81 select starts a fresh transaction. Without this // the memcard stays in mid-protocol after a chip-deselect. mcd.reset(); @@ -313,7 +313,7 @@ pub fn padUpdate(p: *mut[] u8, ic: *mut[] u32, cop0: *mut[] u32, cyc: u32) { return; } padSetU32(p, P_CYC_LO, 0); - // psxe pad.c:329-343: raises IC_JOY unconditionally on countdown + // Raises IC_JOY unconditionally on countdown // expiry, and ONLY sets STAT.IRQ7 (bit 9) if irq_bit was set by // a subsequent-byte TX. The first byte raises IC_JOY without // latching IRQ7 — BIOS sees the ACK pulse but no data-IRQ yet. diff --git a/sdl.jam b/sdl.jam index d27f565..52a0efd 100644 --- a/sdl.jam +++ b/sdl.jam @@ -531,8 +531,8 @@ const BTN_CROSS: u32 = 0x4000; const BTN_SQUARE: u32 = 0x8000; // Translate a USB-HID scancode to a PSX SDA button mask bit. Returns 0 -// for unmapped keys. Keep close to the standard PCSX / DuckStation / -// PSXE keybinding layout so it feels familiar. +// for unmapped keys. Keep close to the standard PCSX / DuckStation +// keybinding layout so it feels familiar. pub fn scancodeToBtn(scan: u32) u32 { match (scan) { SC_UP { return BTN_UP; } diff --git a/sio1.jam b/sio1.jam index d95011b..82d3319 100644 --- a/sio1.jam +++ b/sio1.jam @@ -1,6 +1,5 @@ // SIO1 serial port (link cable / serial-port) at 0x1F801050..0x1F80105F. -// psxe has no real implementation — only two hardcoded reads in bus.c -// (STAT=0x05, CTRL=0). We build the real device per psx-spx so games +// We build the real device per psx-spx so games // that wait on its events (Harvest Moon opens class 0xF0000009 spec 0x20 // = SIO COMMAND_COMPLETE) can complete the SIO transaction loop. // diff --git a/spu.jam b/spu.jam index 8458e61..67fcf7a 100644 --- a/spu.jam +++ b/spu.jam @@ -1,13 +1,13 @@ -// SPU — full port of psxe/psx/dev/spu.c. +// SPU — Sound Processing Unit. // // Layout of the byte buffer returned by spuAlloc: -// 0x000..0x3FF : register file (mirrors psx_spu_t's voice[24] + globals) +// 0x000..0x3FF : register file (the voice[24] register slots + globals) // 0x400.. : 24 × VS_SIZE per-voice runtime data (playing, counter, // current_addr, ADSR state, decoded buf[28], etc.) // 0x1C00.. : global runtime state (taddr, tfifo, even_cycle, revbaddr) // -// Differences from psxe right now: -// - No reverb (the whole spu_get_reverb_sample path is skipped) +// Not yet wired up right now: +// - No reverb (the whole reverb-sample path is skipped) // - No Gaussian interpolation (nearest-neighbor: `out = s[0]`) // - No SDL audio callback yet — spuGetSample is exported but unused // @@ -39,7 +39,7 @@ const OUT_FIFO_CAP: u32 = 4096; const OUT_FIFO_MASK: u32 = 4095; const OUT_FIFO_BYTES: u32 = 16384; // Gaussian-interpolation coefficient table — 512 i16 entries = -// 1024 bytes. Source: psxe spu.c g_spu_gauss_table. Populated by +// 1024 bytes. The standard PSX Gaussian table. Populated by // spuGaussInit at spuAlloc time so we don't need a runtime fopen. const SPU_GAUSS_OFF: u32 = 0x6D00; const SPU_GAUSS_BYTES: u32 = 1024; @@ -85,7 +85,7 @@ const R_CDVOLR: u32 = 0x1B2; // Reverb work-area registers (nocash psx-spx). All i16 values, two // flavours: 'd*' / 'm*' are byte offsets into SPU RAM (multiplied by 8 -// per psxe spu_get_reverb_sample) and 'v*' are i16 fixed-point +// in the reverb-sample path) and 'v*' are i16 fixed-point // coefficients in the range [-0x8000, 0x7FFF]. const R_DAPF1: u32 = 0x1C0; const R_DAPF2: u32 = 0x1C2; @@ -134,7 +134,7 @@ const D_BFLAGS: u32 = 0x24; const D_BUF0: u32 = 0x28; // 28 × i16 → 56 bytes through 0x60 const D_H0: u32 = 0x60; const D_H1: u32 = 0x64; -const D_LVOL: u32 = 0x68; // psxe stores as float; we keep i32 (volumel) +const D_LVOL: u32 = 0x68; // kept as i32 (volumel); a float would also work const D_RVOL: u32 = 0x6C; const D_CVOL: u32 = 0x70; const D_APHASE: u32 = 0x74; @@ -178,7 +178,7 @@ const ADSR_SUSTAIN: u32 = 2; const ADSR_RELEASE: u32 = 3; const ADSR_END: u32 = 4; -// ADPCM filter coefficient tables (psxe spu.c:22-28). +// ADPCM filter coefficient tables. const ADPCM_POS_0: i32 = 0; const ADPCM_POS_1: i32 = 60; const ADPCM_POS_2: i32 = 115; @@ -254,7 +254,7 @@ fn spuStepNoise(s: *mut[] u8, spucnt: u32) { // allocation // Owned SPU-state buffer (~29KB). Initialised via setU16 + spuGaussInit -// (psxe's reset values: all voices ended, irq9addr = 0xFFFF). Auto-drops +// (reset values: all voices ended, irq9addr = 0xFFFF). Auto-drops // with the Bus. pub fn spuAlloc() Vec(u8) { var b: Vec(u8) = Vec(u8).filled(0, SPU_STATE_SIZE); @@ -305,7 +305,7 @@ pub fn sat16i32(v: i32) i32 { } // Multiply 16-bit signed by 16-bit signed coef and shift down by 15 -// (== / 32768). Matches the (vXxx * sample) / 32768 idiom in psxe. +// (== / 32768). This is the standard (vXxx * sample) / 32768 idiom. pub fn mulVol(v: i32, vol: i32) i32 { return (v * vol) >> 15; } @@ -339,12 +339,12 @@ pub fn spuWriteReverb(s: *mut[] u8, ram: *mut[] u8, addr: u32, v: i32) { // Run one reverb sample. inL/inR are the SPU's pre-reverb voice mix. // Returns packed (outL | outR<<16) — both already volume-scaled by // vLOUT / vROUT. Called only when SPUCNT.bit7 is set (reverb master -// enable) and on the alternating "even cycle" — see psxe spu.c line -// 696-697 where get_reverb_sample is gated on spu->even_cycle. +// enable) and on the alternating "even cycle" — the reverb-sample +// computation is gated on the even cycle. pub fn spuGetReverbSample(s: *mut[] u8, ram: *mut[] u8, inL: i32, inR: i32) u32 { // All d*/m* values are <<3 to convert from "halfword index" to - // byte offset (matches psxe's `<< 3`). + // byte offset (the `<< 3`). const dapf1: u32 = (getU16(s, R_DAPF1)) << 3; const dapf2: u32 = (getU16(s, R_DAPF2)) << 3; const mlsame: u32 = (getU16(s, R_MLSAME)) << 3; @@ -540,9 +540,9 @@ pub fn sat16(v: i32) i32 { return v; } -// Wrap an i32 to int16 (low 16 bits, sign-extended). psxe's ADPCM +// Wrap an i32 to int16 (low 16 bits, sign-extended). The ADPCM // reconstruction assigns the sum to an int16_t, which TRUNCATES (wraps) -// rather than saturates (spu.c:165 — its b) { return a; } return b; } -// ADPCM block decoder (psxe spu_read_block) +// ADPCM block decoder pub fn spuReadBlock(s: *mut[] u8, ram: *mut[] u8, v: u32) { const base: u32 = voiceRTBase(v); @@ -581,7 +581,7 @@ pub fn spuReadBlock(s: *mut[] u8, ram: *mut[] u8, v: u32) { if ((nibble & 0x8) != 0) { t = t - 16; } const tShifted: i32 = t << shift; const pred: i32 = ((h0 * f0) + (h1 * f1) + 32) / 64; - // psxe wraps to int16 here (not saturates) — see wrap16. + // wraps to int16 here (not saturates) — see wrap16. const sample: i32 = wrap16(tShifted + pred); h1 = h0; @@ -709,7 +709,7 @@ pub fn spuHandleAdsr(s: *mut[] u8, v: u32) { // Advance every playing voice's ADSR envelope by ONE SPU sample. Called // CPU-synchronously (once per 768 CPU cycles = 44.1 kHz) from runOneFrame // so the envelope level the game POLLS at voice reg 0xC (ENVCVOL) is -// deterministic and matches psxe/hardware — instead of advancing on the +// deterministic and matches hardware — instead of advancing on the // wall-clock audio thread (which desyncs from the CPU and made Brave Fencer // read a wrong envelope level, looping the intro resource-load → black FMV). // The audio thread no longer steps the envelope (removed from spuGetSample). @@ -734,9 +734,9 @@ pub fn spuKon(s: *mut[] u8, ram: *mut[] u8, value: u32) { setU32(s, rt + D_PLAYING, 1); setU32(s, rt + D_CURADDR, getU16(s, vr + VR_ADSADDR) << 3); setU32(s, rt + D_REPADDR, getU16(s, vr + VR_ADRADDR) << 3); - // Store volumes as signed i32 (psxe uses float; the multiplier - // matters more than the type). psxe scales by x2: - // lvol = (volumel / 32767) * 2 (spu.c:342-343). Bake the x2 + // Store volumes as signed i32 (a float would also work; the + // multiplier matters more than the type). The volume scales by x2: + // lvol = (volumel / 32767) * 2. Bake the x2 // in here so the >>15 sample path matches reference amplitude // (without it every voice plays ~6 dB too quiet). setI32(s, rt + D_LVOL, getI16(s, vr + VR_VOLUMEL) * 2); @@ -764,7 +764,7 @@ pub fn spuKoff(s: *mut[] u8, value: u32) { } } -// write-side effects (psxe spu_handle_write) +// write-side effects pub fn spuHandleWrite(s: *mut[] u8, ram: *mut[] u8, off: u32, val: u32) i32 { if (off == R_KONL || off == R_KONH) { @@ -813,7 +813,7 @@ pub fn spuHandleWrite(s: *mut[] u8, ram: *mut[] u8, off: u32, val: u32) i32 { stat = (stat & 0xFFC0) | (val & 0x3F); setU16(s, R_SPUSTAT, stat); // Drain any pending TFIFO data on every SPUCNT write — covers both - // the psxe behavior (flush on entering an active mode) AND the + // a flush on entering an active mode AND the // hardware-spec behavior (flush on transition to Stop). Critically, // this puts the flush BEFORE the next setStartAddress in // setupDMARead, so data lands at the original TADDR (ps1-tests @@ -842,7 +842,7 @@ pub fn spuHandleWrite(s: *mut[] u8, ram: *mut[] u8, off: u32, val: u32) i32 { return 0; } -// read TFIFO (psxe spu_read16 SPUR_TFIFO branch) +// read TFIFO pub fn spuReadTfifo(s: *mut[] u8, ram: *mut[] u8) u32 { var ta: u32 = getU32(s, GLOBAL_BASE + G_TADDR); @@ -878,7 +878,7 @@ pub fn spuWrite8(s: *mut[] u8, ram: *mut[] u8, off: u32, val: u32) { pub fn spuWrite16(s: *mut[] u8, ram: *mut[] u8, off: u32, val: u32) { if (off + 1 >= SPU_REG_SIZE) { return; } if (spuHandleWrite(s, ram, off, val & 0xFFFF) != 0) { return; } - // psxe skips writes at offset 0x0C inside the voice slot (volumer mirror). + // Skip writes at offset 0x0C inside the voice slot (volumer mirror). if ((off & 0xF) == 0xC && off < 0x180) { return; } setU16(s, off, val & 0xFFFF); } @@ -893,7 +893,7 @@ pub fn spuWrite32(s: *mut[] u8, ram: *mut[] u8, off: u32, val: u32) { setU16(s, off + 2, (val >> 16) & 0xFFFF); } -// sample mixer (psxe psx_spu_get_sample) +// sample mixer pub fn spuGetSample(s: *mut[] u8, ram: *mut[] u8, ic: *mut[] u32, cop0: *mut[] u32) u32 { @@ -903,7 +903,7 @@ pub fn spuGetSample(s: *mut[] u8, ram: *mut[] u8, var left: i32 = 0; var right: i32 = 0; - // KON/KOFF are one-shot — psxe zeroes them after dispatch each tick. + // KON/KOFF are one-shot — zero them after dispatch each tick. setU16(s, R_KONL, 0); setU16(s, R_KONH, 0); setU16(s, R_KOFFL, 0); @@ -920,8 +920,8 @@ pub fn spuGetSample(s: *mut[] u8, ram: *mut[] u8, spuStepNoise(s, cnt0); const noiseLvl: i32 = getI16(s, GLOBAL_BASE + G_NOISE_LEVEL); - // Reverb input accumulators — only EON voices feed the reverb (psxe - // spu.c:671-673). prevOut carries the prior voice's ADSR-scaled output + // Reverb input accumulators — only EON voices feed the reverb. + // prevOut carries the prior voice's ADSR-scaled output // for pitch modulation. var revL: i32 = 0; var revR: i32 = 0; @@ -981,7 +981,7 @@ pub fn spuGetSample(s: *mut[] u8, ram: *mut[] u8, } const cur: i32 = getI16(s, rt + D_BUF0 + sampleIdx * 2); setI32(s, rt + D_S0, cur); - // 4-tap Gaussian interpolation. Matches psxe spu.c:642-661. + // 4-tap Gaussian interpolation. const gIdx: u32 = (counter >> 4) & 0xFF; const g0: i32 = spuGaussLookup(s, 0x0FF - gIdx); const g1: i32 = spuGaussLookup(s, 0x1FF - gIdx); @@ -998,10 +998,10 @@ pub fn spuGetSample(s: *mut[] u8, ram: *mut[] u8, const envc: i32 = getI16(s, vr + VR_ENVCVOL); const lvol: i32 = getI32(s, rt + D_LVOL); const rvol: i32 = getI32(s, rt + D_RVOL); - // psxe: samplel = (out * lvol) * (envcvol / 32767) - // i64 fused form: (out * vol * envcvol) >> 30 in one step. psxe - // uses a single float multiply; doing it in i64 avoids the - // precision loss of truncating after an intermediate >>15. + // Conceptually: samplel = (out * lvol) * (envcvol / 32767) + // i64 fused form: (out * vol * envcvol) >> 30 in one step. A + // single float multiply would also work; doing it in i64 avoids + // the precision loss of truncating after an intermediate >>15. const sl: i32 = (((out as i64) * (lvol as i64) * (envc as i64)) >> 30) as i32; const sr: i32 = (((out as i64) * (rvol as i64) * (envc as i64)) >> 30) as i32; left = left + sl; @@ -1060,14 +1060,14 @@ pub fn spuGetSample(s: *mut[] u8, ram: *mut[] u8, // Reverb pass — gated on SPUCNT.bit 7 (REVERB master enable). The // filter chews two samples per turn (the rotating revbaddr advances - // by 2 bytes per call) so psxe only fires it on the even cycle and + // by 2 bytes per call) so it only fires on the even cycle and // caches the result through to the odd cycle. We do the same here: // recompute lrsl/lrsr when even, reuse the cached value otherwise. if ((spucnt & 0x0080) != 0) { const even: u32 = getU32(s, GLOBAL_BASE + G_EVEN_CYC); if (even != 0) { // Reverb input is the EON-voice sum (+ CD when enabled), not the - // full mix — matches psxe spu.c:684-697 and Avocado/DuckStation. + // full mix — matches Avocado/DuckStation. const sl16: i32 = sat16(revL); const sr16: i32 = sat16(revR); const rv: u32 = spuGetReverbSample(s, ram, sl16, sr16); @@ -1158,7 +1158,7 @@ pub fn spuUpdate(s: *mut[] u8, ram: *mut[] u8, // Replaces the previous nearest-neighbor sample lookup with the 4-tap // Gaussian filter PSX hardware uses. Lookup table lives at offset // SPU_GAUSS_OFF inside the SPU state buffer, populated once at -// spuAlloc via spuGaussInit. Per-sample math (matches psxe spu.c:642): +// spuAlloc via spuGaussInit. Per-sample math: // // idx = (counter >> 4) & 0xff // out = (table[0x0FF - idx] * s3 diff --git a/tests.jam b/tests.jam index 6d20e79..c14d644 100644 --- a/tests.jam +++ b/tests.jam @@ -339,7 +339,7 @@ tfn psxLwlLwr() { assert(tLwlLwr(), 0xBBCCDD11); } // GTE (COP2) unit tests // // gteAlloc gives a standalone register buffer, so the math ops can be -// exercised without a full Bus. These lock in the psxe-match fixes: +// exercised without a full Bus. These lock in the GTE fixes: // RTPS stores MAC >> sf (not the raw 44-bit sum), MFC2 sign-extends the // 16-bit registers, and the perspective divide uses the UNR LUT. @@ -451,7 +451,7 @@ fn tGteMvmvaNestedOvf() u32 { // IRGB/ORGB (reg 28) repacks IR1/2/3 (>>7, clamped 0..0x1F) into 15-bit RGB. // IR is stored zero-extended in the low 16 bits and must be sign-extended -// before the >>7, so a negative IR clamps to 0 (psxe cpu.c:1290 / duckstation +// before the >>7, so a negative IR clamps to 0 (DuckStation // gte.cpp:343). The old `(g[9] as i32)` bit-cast read IR1=-1 as +65535 → 0x1F; // with all three IR = -1 the packed result must be 0, not 0x7FFF. fn tGteIrgbNeg() u32 { @@ -472,8 +472,8 @@ fn tGteIrgbPos() u32 { return gteDataRead(g.ptr, 28); } -// RTPT runs the depth-cue (DQ) tail only on the LAST vertex (psxe -// cpu.c:2247-2251). Construct V0/V1 with a small Z (→ large divide → the DQ's +// RTPT runs the depth-cue (DQ) tail only on the LAST vertex. +// Construct V0/V1 with a small Z (→ large divide → the DQ's // IR0 = clamp(DQA*div>>12) saturates, setting FLAG bit 12) and V2 with a large // Z (→ small divide → no IR0 saturation). Bit 12 (IR0 sat) is set ONLY by the // DQ tail's gte_clamp_ir0 and is excluded from the bit-31 summary, so it @@ -492,8 +492,8 @@ fn tGteRtptDqGate() u32 { return gteCtrlRead(g.ptr, 31) & 0x1000; // FLAG bit 12 = IR0 saturation (DQ tail only) } -// NCDS pushes the final colour via the FLAG-aware gte_clamp_rgb (psxe -// cpu.c:1917-1919), so a channel that saturates >255 sets FLAG bit 21/20/19 +// NCDS pushes the final colour via the FLAG-aware RGB clamp, +// so a channel that saturates >255 sets FLAG bit 21/20/19 // (R/G/B). Drive only the R far-colour large and IR0 to max so stage-5 // MAC1 = IR0*ir1f hugely overflows 255 on R alone; bits 19/20/21 masked must // read 0x200000 (R). The old inline clampU8 set none of these (returned 0). @@ -505,7 +505,7 @@ fn tGteNcdsRgbSat() u32 { return gteCtrlRead(g.ptr, 31) & 0x00380000; // RGB-saturation FLAG bits 19/20/21 } -// DQA (control reg 27) is s16 (psxe cpu.c:138 / duckstation gte_types.h:117), +// DQA (control reg 27) is s16 (DuckStation gte_types.h:117), // so the RTPS depth-cue must sign-extend its low 16 bits, not read all 32. With // DQA=0x00010001 the low half is 1: depth-cue MAC0 = DQB + DQA*div = 1*0x10000, // IR0 = clamp(0x10000>>12) = 16. The old s32 read used 65537, giving a huge @@ -667,14 +667,14 @@ tfn emuDiscBootsExec() { // pushes an INT3/INT2/INT5 response, so a non-zero value proves the CD // state machine got exercised. -// variable per-instruction cycle model (psxe cpu.c last_cycles) +// variable per-instruction cycle model // -// step() charges `fetchCyc + instrCyc` per instruction, matching psxe: -// fetchCyc = 18 when the opcode is read from BIOS ROM (bios.c bus_delay) +// step() charges `fetchCyc + instrCyc` per instruction: +// fetchCyc = 18 when the opcode is read from BIOS ROM (BIOS bus delay) // else 0; instrCyc = 2 except COP2 math ops, which cost their GTE op time. -// These deltas drive device timing (CDROM/DMA/timers) at psxe's rate. +// These deltas drive device timing (CDROM/DMA/timers) at the right rate. -// gteOpCycles mirrors psxe's GTE cycle switch (cpu.c). Low 6 bits select. +// gteOpCycles selects the GTE op cycle count from the low 6 bits. fn tGteCycRtps() u32 { return gteOpCycles(0x4A000001); } // RTPS → 15 fn tGteCycNcdt() u32 { return gteOpCycles(0x4A000016); } // NCDT → 44 fn tGteCycRtpt() u32 { return gteOpCycles(0x4A000030); } // RTPT → 23 diff --git a/timer.jam b/timer.jam index 951f90a..ab55557 100644 --- a/timer.jam +++ b/timer.jam @@ -46,8 +46,8 @@ pub const Timer = struct { vblank: u32, channels: [3]TimerChannel, // Timer 0 dot-clock ticks per sysclk cycle = (11/7)/hdiv, refreshed from - // the GPU display mode each frame (setDotclock). Mirrors psxe - // timer_get_dotclock_div. Default = 320-wide (hdiv 8). + // the GPU display mode each frame (setDotclock). + // Default = 320-wide (hdiv 8). dotDiv: f32, pub fn init() Self { @@ -64,8 +64,7 @@ pub const Timer = struct { // Refresh the Timer 0 dot-clock divisor from the GPU display mode // (GP1 08h, low 24 bits). hdiv table {256:10, 320:8, 512:5, 640:4}, - // or 7 for the 368-wide (bit 6) mode — same as psxe's - // dmode_dotclk_div_table. + // or 7 for the 368-wide (bit 6) mode. pub fn setDotclock(self: mut Self, dm: u32) { if ((dm & 0x40) != 0) { self.dotDiv = 11.0 / 7.0 / 7.0; @@ -79,8 +78,8 @@ pub const Timer = struct { self.dotDiv = 11.0 / 7.0 / div; } - // Compose the mode register from individual flag fields, per - // psxe `timer_get_mode()`. Side effect: clears + // Compose the mode register from individual flag fields. + // Side effect: clears // `targetReached` and `maxReached` (they latch until read). pub fn getMode(self: mut Self, idx: u32) u32 { var v: u32 = 0; @@ -101,8 +100,8 @@ pub const Timer = struct { } // Mode write — decompose the value into the per-flag fields, - // then reset side-state (counter, irq, fired, etc.) per psxe - // semantics. Also seeds `paused` from hblank/vblank for sync + // then reset side-state (counter, irq, fired, etc.). + // Also seeds `paused` from hblank/vblank for sync // modes. pub fn setMode(self: mut Self, idx: u32, val: u32) { self.channels[idx].syncEnable = (val >> 0) & 1; @@ -153,11 +152,10 @@ pub const Timer = struct { // on target-hit if resetTarget is set. pub fn handleIrq(self: mut Self, idx: u32, ic: *mut[] u32, cop0: *mut[] u32) { - // psxe timer_handle_irq compares the FLOAT counter directly to the - // target and to 65535.0f with strict `>` (timer.c:236-237). jam had - // used `>=` (toward hardware/duckstation), but that fires the timer - // IRQ one tick earlier than psxe on every exact landing — a per-IRQ - // delivery-timing divergence from the FMV's parity reference. Match psxe. + // Compare the FLOAT counter directly to the target and to 65535.0f + // with strict `>`. Using `>=` (toward hardware/DuckStation) would + // fire the timer IRQ one tick earlier on every exact landing — a + // per-IRQ delivery-timing divergence. Stick with `>`. const counter: f32 = self.channels[idx].counter; const target: u32 = self.channels[idx].target; var fireIrq: u32 = 0; @@ -258,13 +256,12 @@ pub const Timer = struct { pub fn updateCyc(self: mut Self, ic: *mut[] u32, cop0: *mut[] u32, cyc: u32) { - // psxe's psx_timer_update (timer.c:338-345) IGNORES its cyc arg and - // advances all three timers by a hardcoded 2 per instruction — - // NOT the real per-instruction cycle count. (timer_update_timer0/1/2 - // do `+= (float)cyc`, but cyc is always 2.) Feeding the real `cyc` - // (fetchCyc+2 = 20 for BIOS-resident code) made jam's sysclk timers - // run ~10× too fast in BIOS/kernel loops, drifting any timer value the - // game reads. Match psxe: pass 2. (`cyc` kept for signature parity.) + // IGNORE the cyc arg and advance all three timers by a hardcoded 2 + // per instruction — NOT the real per-instruction cycle count. + // Feeding the real `cyc` (fetchCyc+2 = 20 for BIOS-resident code) + // makes the sysclk timers run ~10× too fast in BIOS/kernel loops, + // drifting any timer value the game reads. Pass 2. (`cyc` kept for + // signature parity.) self.updateOne(0, 2, ic, cop0); self.updateOne(1, 2, ic, cop0); self.updateOne(2, 2, ic, cop0); @@ -276,14 +273,13 @@ pub const Timer = struct { // Timer 1 sourced from hblank is incremented in hblankBegin. if (idx == 1 && (self.channels[idx].clkSource & 1) != 0) { return; } const cf: f32 = cyc as f32; - // Mirror psxe timer_update_timer* (timer.c:300-333) on the float - // counter. Timer 2 with clk_source>=2 ticks at sysclk/8 (`+= cyc/8`). - // Timer 0 with clk_source bit0 set ticks at the GPU dot-clock - // (`+= cyc * dotDiv`, dotDiv = (11/7)/hdiv) instead of full sysclk — - // matches all four reference emulators (the dot-clock is slower than - // sysclk, so the old `+= cyc` ran T0 several× too fast). The old jam - // integer subtick + leap-past-target adjustment is dropped: psxe does - // neither, and matching its exact float overshoot/reset is the point. + // Advance the float counter. Timer 2 with clk_source>=2 ticks at + // sysclk/8 (`+= cyc/8`). Timer 0 with clk_source bit0 set ticks at + // the GPU dot-clock (`+= cyc * dotDiv`, dotDiv = (11/7)/hdiv) instead + // of full sysclk — matches the reference emulators (the dot-clock is + // slower than sysclk, so a plain `+= cyc` would run T0 several× too + // fast). No integer subtick or leap-past-target adjustment: the float + // overshoot/reset is handled directly by handleIrq. if (idx == 2 && self.channels[idx].clkSource >= 2) { self.channels[idx].counter = self.channels[idx].counter + cf / 8.0; } else if (idx == 0 && (self.channels[idx].clkSource & 1) != 0) { @@ -301,7 +297,7 @@ pub const Timer = struct { if ((self.channels[1].clkSource & 1) != 0 && self.channels[1].paused == 0) { - // psxe increments the hblank-sourced T1 counter by 1.0 each + // Increment the hblank-sourced T1 counter by 1.0 each // hblank; handleIrq's float overflow/target checks reset it. self.channels[1].counter = self.channels[1].counter + 1.0; self.handleIrq(1, ic, cop0); diff --git a/xa.jam b/xa.jam index a2b366e..d4fb6dd 100644 --- a/xa.jam +++ b/xa.jam @@ -48,7 +48,7 @@ pub fn xaSectorIsAudio(secBuf: *mut[] u8) bool { } // Form-2 bit (sub-mode byte bit 0) — form-2 sectors don't have ECC and -// always carry audio/video. psxe fetch_xa_sector skips non-form-2 sectors +// always carry audio/video. The XA fetch path skips non-form-2 sectors // even when MODE_XA_ADPCM is set. pub fn xaSectorIsForm2(secBuf: *mut[] u8) bool { return (secBuf[0x12] & 0x01) != 0; @@ -79,8 +79,9 @@ pub fn xaSector4Bit(secBuf: *mut[] u8) bool { } // Decode one 28-sample block. `idx` is the byte offset to the start of -// the 128-byte sound group inside the sector buffer (psxe walks idx=24, -// 152, 280, ...). `blk` is the block index within the group (0..3). +// the 128-byte sound group inside the sector buffer (idx walks 24, +// 152, 280, ... in steps of 128). `blk` is the block index within the +// group (0..3). // `nib` is the nibble half (0 = low, 1 = high). `dst` receives 28 i16 // samples; `hist` is a 2-entry signed-16 history (hist[0] = previous // sample, hist[1] = sample-before-previous). @@ -88,13 +89,14 @@ pub fn xaDecodeBlock(secBuf: *mut[] u8, idx: u32, blk: u32, nib: u32, dst: *mut[] i32, hist: *mut[] i32) { const hdr: u32 = secBuf[idx + 4 + blk * 2 + nib] as u32; - // Reserved ADPCM shift values 13..15 act as 9 on hardware (duckstation - // cdrom.cpp:285-289). psxe's naive `12 - (hdr&0xF)` leaves shift negative - // and the `old << shift` below is then an undefined negative shift; clamp. + // Reserved ADPCM shift values 13..15 act as 9 on hardware (DuckStation + // cdrom.cpp:285-289). A naive `12 - (hdr&0xF)` would leave shift negative + // and the `old << shift` below would then be an undefined negative shift; + // clamp. var shiftVal: i32 = (hdr & 0x0F) as i32; if (shiftVal > 12) { shiftVal = 9; } const shift: i32 = 12 - shiftVal; - // self note: psxe masks to 0x30 then >>4 + // self note: mask to 0x30 then >>4 const filter: u32 = (hdr >> 4) & 3; const f0: i32 = xaPosCoef(filter); const f1: i32 = xaNegCoef(filter); @@ -108,7 +110,7 @@ pub fn xaDecodeBlock(secBuf: *mut[] u8, const old: i32 = (t << 12) >> 12; const shifted: i32 = old << shift; const pred: i32 = (f0 * hist[0] + f1 * hist[1] + 32) / 64; - // self note: psxe wraps to int16 here, not saturates + // self note: wrap to int16 here, do not saturate const s: i32 = xaWrap16(shifted + pred); hist[1] = hist[0]; hist[0] = s; -- 2.51.2