diff --git a/Makefile b/Makefile index 01931d6..17f962c 100644 --- a/Makefile +++ b/Makefile @@ -9,10 +9,15 @@ ifeq ($(UNAME_S),Darwin) SDL_LIB ?= /opt/homebrew/lib SDL_INC ?= /opt/homebrew/include RUN_ENV := DYLD_LIBRARY_PATH=$(SDL_LIB) + # libobjc: sdl.jam's disableAppNap() calls the ObjC runtime to switch off + # macOS App Nap (which otherwise stalls the emulator when the window is + # unfocused). The call is @isDarwin()-guarded, so -lobjc is Darwin-only. + OBJC_LIB := -lobjc else SDL_LIB ?= /usr/lib SDL_INC ?= /usr/include RUN_ENV := LD_LIBRARY_PATH=$(SDL_LIB) + OBJC_LIB := endif # Point at the std/ directory in the in-tree jam build. The installed @@ -27,10 +32,10 @@ test: $(BUILD_ENV) $(JAM) test tests.jam build: - $(BUILD_ENV) $(JAM) -lSDL2 -o $(BIN) main.jam + $(BUILD_ENV) $(JAM) -lSDL2 $(OBJC_LIB) -o $(BIN) main.jam release: - $(BUILD_ENV) $(JAM) -C opt-level=3 -lSDL2 -o $(BIN) main.jam + $(BUILD_ENV) $(JAM) -C opt-level=3 -lSDL2 $(OBJC_LIB) -o $(BIN) main.jam run: build $(RUN_ENV) ./$(BIN) diff --git a/cdrom.jam b/cdrom.jam index 974088d..c1ba324 100644 --- a/cdrom.jam +++ b/cdrom.jam @@ -316,12 +316,14 @@ pub const Cdrom = struct { var delay: u32 = CD_DELAY_FR; if ((val & 0xFF) == CDL_INIT) { delay = CD_DELAY_INIT_FR; } self.delay = delay; - // psxe cdrom.c:668-669: a command issued while a read is in - // progress raises BUSYSTS (status bit 7) until the command's - // first response is ready. Without this, a game whose CD-sync - // polls BUSYSTS during streaming never sees the busy edge and - // spins re-issuing SetMode/SetLoc/SeekL/ReadN (FMV hang). - if (self.prevState == CD_STATE_READ as u8) { self.busy = 1; } + // psxe's cdrom_write_cmd checks `if (state == CD_STATE_READ) busy=1` + // (cdrom.c:683) but `state` was just set to TX_RESP1 two lines above + // (cdrom.c:669), so that branch is ALWAYS FALSE — psxe never raises + // BUSYSTS from a command write. We must not either: asserting busy on + // the mid-read GetlocL/Getstat polls that the BFM/Suikoden FMV loader + // spams makes the game see the drive busy where psxe shows it idle, so + // it re-reads the resource sectors forever and never triggers the MDEC + // decode (FMV black-screen). } // Push parameter byte. 2 { self.pushParam(val); } @@ -483,7 +485,11 @@ pub const Cdrom = struct { self.ifr = 3; self.pushResp(self.getStat()); self.state = CD_STATE_TX_RESP2 as u8; - self.delay = CD_DELAY_INIT_FR; + // psxe cdrom_cmd_init (impl.c:308) sets the 2nd-response delay to + // CD_DELAY_1MS (33869), NOT CD_DELAY_INIT_FR (81102). The 47233-cyc + // gap made jam's Init INT2 fire late, slipping the CD IRQ at ~164.8M + // (the next trace divergence after the GPU-acc/-ffast-math one). + self.delay = CD_DELAY_1MS; return; } if (cmd == CDL_SEEKL) { @@ -855,20 +861,24 @@ pub const Cdrom = struct { // command in TX_RESP1 incurs a delay. if (self.state == CD_STATE_READ as u8) { self.processSetloc(); - // First sector after a command: add psxe's pending - // speed-switch resync (0 unless SetMode just flipped 1x<->2x), - // consumed once. Normal reads are unchanged. - self.delay = self.readDelay() + 4 * 33869 + self.pendingSpeedSwitch; + // First sector after a command uses psxe's CD_DELAY_ONGOING_READ + // (= readDelay + 4ms). psxe COMPUTES a speed-switch resync delay + // but then OVERWRITES it with CD_DELAY_ONGOING_READ (cdrom.c:567), + // so it never applies; adding pendingSpeedSwitch (~650ms once) + // delayed the first FMV sector vs psxe. Drop it; just consume it. + self.delay = self.readDelay() + 4 * 33869; self.pendingSpeedSwitch = 0; } } else if (self.state == CD_STATE_TX_RESP2 as u8) { self.executeResp2(disc); if (self.state == CD_STATE_READ as u8) { self.processSetloc(); - // First sector after a command: add psxe's pending - // speed-switch resync (0 unless SetMode just flipped 1x<->2x), - // consumed once. Normal reads are unchanged. - self.delay = self.readDelay() + 4 * 33869 + self.pendingSpeedSwitch; + // First sector after a command uses psxe's CD_DELAY_ONGOING_READ + // (= readDelay + 4ms). psxe COMPUTES a speed-switch resync delay + // but then OVERWRITES it with CD_DELAY_ONGOING_READ (cdrom.c:567), + // so it never applies; adding pendingSpeedSwitch (~650ms once) + // delayed the first FMV sector vs psxe. Drop it; just consume it. + self.delay = self.readDelay() + 4 * 33869; self.pendingSpeedSwitch = 0; } } else if (self.state == CD_STATE_READ as u8) { diff --git a/cpu.jam b/cpu.jam index 293e928..0c37188 100644 --- a/cpu.jam +++ b/cpu.jam @@ -59,10 +59,6 @@ const Cpu = struct { branchTaken: u8, halted: u8, cycles: u64, - // DIAG: one-shot dump fields for IRQ-handler-chain investigation. - diagDumped: u32, - diagExcCount: u32, - diagWalkCount: u32, }; // COP0 register indices. @@ -118,7 +114,6 @@ pub fn freshCpu() Cpu { loadD: 0, loadV: 0, branch: 0, delaySlot: 0, branchTaken: 0, halted: 0, cycles: 0, - diagDumped: 0, diagExcCount: 0, diagWalkCount: 0, }; return c; } diff --git a/gpu.jam b/gpu.jam index bb5781e..3e8e91a 100644 --- a/gpu.jam +++ b/gpu.jam @@ -1,11 +1,9 @@ // GPU (partial port of psxe/psx/dev/gpu.c). // -// Implements the GP0/GP1 command machinery, plus the rectangle and -// CPU↔VRAM blit commands the BIOS uses during boot. Polygon and line -// rasterizers are stubbed (they consume args without drawing); games -// won't render properly until a triangle rasterizer + GTE land. The -// goal of this port is to keep the BIOS from hanging while waiting on -// GPUSTAT and to write something visible to VRAM. +// Implements the GP0/GP1 command machinery and the polygon, line, +// rectangle, fill, and CPU↔VRAM blit commands. Triangles are +// point-sampled with texture + gouraud support (gpuRasterPoly → +// gpuRasterTri); together with the GTE this renders real game geometry. // // State buffer layout (80 u32 entries): // [0..15] command FIFO buffer @@ -17,7 +15,10 @@ // [21] color (current command's 24-bit BGR colour) // [22..29] generic command counters: xpos, ypos, xsiz, ysiz, tsiz, // addr, xcnt, ycnt -// [30..37] v0..v3 (x,y per vertex) +// [30..33] unused (was v0..v3 vertex scratch; the rasterizer reads none) +// [34..37] OWNED BY main.jam's frame loop, NOT the GPU: [34]=last CPU-clocked +// SPU sample, [35]=SPU sample accumulator, [36]=scanline, +// [37]=f32 GPU-cycle accumulator (type-punned). Do NOT reuse here. // [38..43] C0 (VRAM→CPU) state: xcnt, ycnt, xsiz, ysiz, tsiz, addr // [44] gp1_10h_req // [45] gpuread @@ -29,8 +30,9 @@ // [58..59] clut_x, clut_y // [60..62] texp_x, texp_y, texp_d // [63..68] disp_x, disp_y, disp_x1..disp_y2 -// [69] cycles (GPU clock cycles within current scanline) -// [70] line (current scanline) +// [69] unused +// [70] texture-disable raw input (E1 bit 11) +// [71] texture-disable allow gate (GP1 0x09); GPUSTAT.15 = g[71] & g[70] const { irqRaise, IC_VBLANK } = import("irq"); @@ -567,81 +569,9 @@ pub fn gpuMax3(a: u32, b: u32, c: u32) u32 { return m; } -// Rasterize one flat-shaded triangle. Caller has already applied the -// drawing offset to the vertex coordinates. Pixels are clipped to the -// drawing area registered via GP0 0xE3 / 0xE4. -pub fn gpuFlatTriangle(g: *mut[] u32, vram: *mut[] u8, - a0x: u32, a0y: u32, - b0x: u32, b0y: u32, - c0x: u32, c0y: u32, - color: u32) { - // Apply drawing offset (sign-extended in GP0 0xE5). - const ox: u32 = g[52]; - const oy: u32 = g[53]; - const ax: u32 = a0x + ox; - const ay: u32 = a0y + oy; - var bx: u32 = b0x + ox; - var by: u32 = b0y + oy; - var cx: u32 = c0x + ox; - var cy: u32 = c0y + oy; - - // Force CCW winding: if the triangle is CW, swap b and c. - const area: i64 = gpuEdge(ax, ay, bx, by, cx, cy); - if (area < 0) { - const tx: u32 = bx; const ty: u32 = by; - bx = cx; by = cy; - cx = tx; cy = ty; - } - - const xmin: u32 = gpuMin3(ax, bx, cx); - const ymin: u32 = gpuMin3(ay, by, cy); - const xmax: u32 = gpuMax3(ax, bx, cx); - const ymax: u32 = gpuMax3(ay, by, cy); - - // Clip to drawing area. - const dx1: u32 = g[48]; - const dy1: u32 = g[49]; - const dx2: u32 = g[50]; - const dy2: u32 = g[51]; - - // Bail on pathologically large bounding boxes (PSX hardware caps). - if (sext32(xmax - xmin) > 1024 || sext32(ymax - ymin) > 512) { return; } - - // Use signed comparisons throughout. The drawing-area registers - // are non-negative, so an unsigned `x >= dx1` would erroneously - // include far-negative x positions (sext32 0xFFFFF800 > 0 in - // unsigned terms). - var y: u32 = ymin; - while (sext32(y) <= sext32(ymax)) { - var x: u32 = xmin; - while (sext32(x) <= sext32(xmax)) { - if (sext32(x) >= sext32(dx1) && sext32(x) <= sext32(dx2) && - sext32(y) >= sext32(dy1) && sext32(y) <= sext32(dy2)) { - const z0: i64 = gpuEdge(bx, by, cx, cy, x, y); - const z1: i64 = gpuEdge(cx, cy, ax, ay, x, y); - const z2: i64 = gpuEdge(ax, ay, bx, by, x, y); - if (!gpuTopLeftRule(z0, bx, by, cx, cy) && - !gpuTopLeftRule(z1, cx, cy, ax, ay) && - !gpuTopLeftRule(z2, ax, ay, bx, by)) { - vramWritePixel(vram, x & 0x3FF, y & 0x1FF, color); - } - } - x = x + 1; - } - y = y + 1; - } -} - // Polygon dispatcher — reads the command flags and triggers one or // two triangles. Handles monochrome and textured variants; gouraud // shading falls back to flat-shaded with the base colour. -// Decode a shaded polygon's per-vertex colour. For shaded prims each -// vertex has an RGB byte triple stored in the low 24 bits of a buffer -// word; we extract by index. -pub fn gpuShadeColor(g: *mut[] u32, idx: u32) u32 { - return g[idx] & 0xFFFFFF; -} - pub fn gpuRasterPoly(g: *mut[] u32, vram: *mut[] u8) { const flags: u32 = (g[0] >> 24) & 0xFF; const isQuad: bool = (flags & 0x08) != 0; diff --git a/gte.jam b/gte.jam index 166a361..4fe9485 100644 --- a/gte.jam +++ b/gte.jam @@ -1150,13 +1150,6 @@ pub fn gteNcds(g: *mut[] u32, vidx: u32, sf: u32, lm: u32) { gtePushRgb(g, m1c, m2c, m3c); } -// 8-bit unsigned clamp used by the RGB FIFO writes. -pub fn clampU8(v: i64) u32 { - if (v < 0) { return 0; } - if (v > 255) { return 255; } - return (v as u64 & 0xFF) as u32; -} - // SQR — square each IR component. Used for vector-length computations. // psxe: MAC{1,2,3} = IR{1,2,3}^2; IR{1,2,3} = clamp(MAC). pub fn gteSqr(g: *mut[] u32, sf: u32, lm: u32) { diff --git a/main.jam b/main.jam index 5fad58e..6c9d2b1 100644 --- a/main.jam +++ b/main.jam @@ -98,9 +98,6 @@ const Cpu = struct { branchTaken: u8, halted: u8, cycles: u64, - diagDumped: u32, - diagExcCount: u32, - diagWalkCount: u32, }; const Sdl = struct { @@ -333,7 +330,6 @@ fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, bus.timer.hblankEnd(); acc = acc - GPU_CYCLES_SCANL; } - // 3. PAD (psx.c:97) padUpdate(bus.pad.ptr, bus.irq.ptr, cop0, delta); // 4. TIMER sysclk (psx.c:98) — updateCyc ignores delta, uses 2 (psxe). @@ -697,17 +693,17 @@ fn main() { // polling runs at VBlank cadence in the BIOS. padSetButtons(bus.pad.ptr, buttons[0]); - // Pace this frame to ~59.94 Hz. The VSync present above already - // blocked to a host vblank; pad the rest of the frame so the loop - // runs at the PSX rate rather than the (possibly 120 Hz) panel rate. + // Pace this frame to ~59.94 Hz. The renderer runs WITHOUT vsync (see + // sdlInit — vsync would block the present on an unfocused window), so + // this wall-clock limiter is the sole pacer: sleep out the rest of the + // 16.68 ms frame so the loop runs at the PSX rate, not flat-out. nextFrameMs = nextFrameMs + FRAME_MS; const nowMs: u32 = SDL_GetTicks(); if ((nowMs as f64) < nextFrameMs) { SDL_Delay((nextFrameMs - (nowMs as f64)) as u32); } else { - // Fell behind (slow frame, or a 60 Hz host where VSync already - // paced us) — resync so we never burst-catch-up faster than - // real time. + // Fell behind (slow frame) — resync so we never burst-catch-up + // faster than real time. nextFrameMs = nowMs as f64; } } diff --git a/sdl.jam b/sdl.jam index e606cf8..289c205 100644 --- a/sdl.jam +++ b/sdl.jam @@ -43,11 +43,14 @@ const SDL_KEYUP: u32 = 0x301; // SDL2 surface — pointer-opaque handles, treated as `*mut[] u8` from Jam. extern fn SDL_Init(flags: u32) i32; extern fn SDL_Quit(); +extern fn SDL_SetHint(name: *const[] u8, value: *const[] u8) i32; extern fn SDL_CreateWindow(title: *const[] u8, x: i32, y: i32, w: i32, h: i32, flags: u32) *mut[] u8; extern fn SDL_DestroyWindow(window: *mut[] u8); extern fn SDL_CreateRenderer(window: *mut[] u8, index: i32, flags: u32) *mut[] u8; +extern fn SDL_GetRendererInfo(renderer: *mut[] u8, info: *mut[] u8) i32; +extern fn strlen(s: *const[] u8) u64; extern fn SDL_DestroyRenderer(renderer: *mut[] u8); extern fn SDL_CreateTexture(renderer: *mut[] u8, format: u32, access: i32, w: i32, h: i32) *mut[] u8; @@ -62,6 +65,17 @@ extern fn SDL_Delay(ms: u32); extern fn SDL_PollEvent(event: *mut[] u8) i32; extern fn SDL_GetTicks() u32; +// --- macOS App Nap disable (Objective-C runtime) --- +// When the jam window is not focused, macOS App Nap throttles the process's +// timers + CPU, stalling the single-threaded emulation loop ("stuck when not +// focused"). Disable it by holding a user-initiated NSProcessInfo activity. +// objc_msgSend is declared NON-variadic with fixed u64 args so every arg lands +// in a register (the ARM64 ABI passes variadic args on the stack, which would +// break objc_msgSend); unused trailing args are ignored by the target method. +extern fn objc_getClass(name: *const[] u8) u64; +extern fn sel_registerName(name: *const[] u8) u64; +extern fn objc_msgSend(recv: u64, sel: u64, a: u64, b: u64) u64; + // --- audio --- // // Callback-mode SDL audio. SDL drives the SPU clock: every ~13 ms @@ -111,8 +125,38 @@ const Sdl = struct { // safe across SDL releases. const SDL_EVENT_SIZE: u32 = 64; +// Hold a process-lifetime "user initiated" NSProcessInfo activity, which +// disables macOS App Nap so the emulator keeps running at full speed when its +// window is unfocused/background. The activity token is intentionally leaked +// (releasing it would re-enable App Nap). +fn disableAppNap() { + // Guarded by @isDarwin() so non-macOS builds dead-code-eliminate the body + // and never reference the objc_* symbols (no -lobjc needed off Darwin), + // mirroring std/process.jam's _NSGetArgv pattern. + if (@isDarwin()) { + var nmPi: []u8 = "NSProcessInfo"; + const clsPi: u64 = objc_getClass(nmPi.ptr); + if (clsPi == 0) { return; } + var selPi: []u8 = "processInfo"; + const pi: u64 = objc_msgSend(clsPi, sel_registerName(selPi.ptr), 0, 0); + if (pi == 0) { return; } + var nmStr: []u8 = "NSString"; + var selStr: []u8 = "stringWithUTF8String:"; + var reason: []u8 = "jamstation emulation running"; + const nsstr: u64 = objc_msgSend(objc_getClass(nmStr.ptr), + sel_registerName(selStr.ptr), reason.ptr as u64, 0); + // NSActivityUserInitiated (0x00FFFFFF) → App Nap disabled while held. + var selBegin: []u8 = "beginActivityWithOptions:reason:"; + const token: u64 = objc_msgSend(pi, sel_registerName(selBegin.ptr), + 0x00FFFFFF, nsstr); + var selRetain: []u8 = "retain"; + objc_msgSend(token, sel_registerName(selRetain.ptr), 0, 0); + } +} + pub fn sdlInit(title: []u8, width: i32, height: i32) Sdl { SDL_Init(SDL_INIT_VIDEO | SDL_INIT_EVENTS | SDL_INIT_AUDIO); + disableAppNap(); // keep running at full speed when the window is unfocused // Window is texture-sized (no upscale) and resizable so the user // can scale up post-hoc. SDL_RenderCopy stretches the texture to // fit whatever the current window dimensions are. @@ -120,8 +164,38 @@ pub fn sdlInit(title: []u8, width: i32, height: i32) Sdl { 0x2FFF0000, 0x2FFF0000, width, height, SDL_WINDOW_SHOWN | SDL_WINDOW_RESIZABLE); - var ren: *mut[] u8 = SDL_CreateRenderer(win, -1, - SDL_RENDERER_ACCELERATED | SDL_RENDERER_PRESENTVSYNC); + // Force the OpenGL renderer instead of macOS's default Metal one. Metal's + // SDL_RenderPresent calls [CAMetalLayer nextDrawable], which BLOCKS when + // the window isn't being composited (unfocused/occluded) because the + // drawable pool starves — this is the "stuck when not focused" freeze, and + // it happens even with vsync off. OpenGL's swap has no such pool, so the + // loop keeps running when unfocused. We also drop PRESENTVSYNC so the + // present never waits on a vblank (which the OS may stop delivering to an + // unfocused window); jam's own 59.94 Hz SDL_Delay limiter paces the loop. + var hintName: []u8 = "SDL_RENDER_DRIVER"; + var hintGl: []u8 = "opengl"; + SDL_SetHint(hintName.ptr, hintGl.ptr); + var ren: *mut[] u8 = SDL_CreateRenderer(win, -1, SDL_RENDERER_ACCELERATED); + if (ren as u64 == 0) { + // OpenGL unavailable — clear the hint and fall back to the default + // (Metal) renderer so we still get a window rather than crashing. + var hintEmpty: []u8 = ""; + SDL_SetHint(hintName.ptr, hintEmpty.ptr); + ren = SDL_CreateRenderer(win, -1, SDL_RENDERER_ACCELERATED); + } + // Report the live render driver so it's obvious which backend won (if this + // says "metal", the OpenGL hint didn't take and the unfocused freeze will + // return). SDL_RendererInfo starts with a `const char *name`; pad to 128 B. + var info: [128]u8 = [0; 128]; + SDL_GetRendererInfo(ren, info.asMutPtr()); + var infoWords: *mut[] u64 = info.asMutPtr() as *mut[] u64; + const namePtr: u64 = infoWords[0]; + if (namePtr != 0) { + const np: *const[] u8 = namePtr as *const[] u8; + const nlen: u64 = strlen(np); + const nameSlice: []u8 = np[0..nlen]; + print("[sdl] render driver: {nameSlice}\n"); + } // Use ARGB8888 (32-bit) so the blit can target both 15bpp and 24bpp // PSX display modes through a single texture. The conversion happens // per-pixel in sdlBlit; SDL gets a canonical 4-byte format. diff --git a/tests.jam b/tests.jam index b8e1f27..67f7071 100644 --- a/tests.jam +++ b/tests.jam @@ -62,9 +62,6 @@ const Cpu = struct { branchTaken: u8, halted: u8, cycles: u64, - diagDumped: u32, - diagExcCount: u32, - diagWalkCount: u32, }; @@ -477,6 +474,79 @@ fn tGteMvmvaNestedOvf() u32 { return f; } +// IRGB/ORGB (reg 28) repacks IR1/2/3 (>>7, clamped 0..0x1F) into 15-bit RGB. +// IR is stored zero-extended in the low 16 bits and must be sign-extended +// before the >>7, so a negative IR clamps to 0 (psxe cpu.c:1290 / duckstation +// gte.cpp:343). The old `(g[9] as i32)` bit-cast read IR1=-1 as +65535 → 0x1F; +// with all three IR = -1 the packed result must be 0, not 0x7FFF. +fn tGteIrgbNeg() u32 { + var g: Vec(u32) = gteAlloc(); + gteDataWrite(g.ptr, 9, 0x0000FFFF); // IR1 = -1 + gteDataWrite(g.ptr, 10, 0x0000FFFF); // IR2 = -1 + gteDataWrite(g.ptr, 11, 0x0000FFFF); // IR3 = -1 + return gteDataRead(g.ptr, 28); // each channel clamps to 0 → 0 +} + +// Positive packing still works after the sign-extension fix: IR1=3968→31, +// IR2=128→1, IR3=0 ⇒ 0x1F | (1<<5) | 0 = 0x3F. +fn tGteIrgbPos() u32 { + var g: Vec(u32) = gteAlloc(); + gteDataWrite(g.ptr, 9, 0x0F80); // IR1 = 3968 → >>7 = 31 + gteDataWrite(g.ptr, 10, 0x0080); // IR2 = 128 → >>7 = 1 + gteDataWrite(g.ptr, 11, 0x0000); // IR3 = 0 + return gteDataRead(g.ptr, 28); +} + +// RTPT runs the depth-cue (DQ) tail only on the LAST vertex (psxe +// cpu.c:2247-2251). Construct V0/V1 with a small Z (→ large divide → the DQ's +// IR0 = clamp(DQA*div>>12) saturates, setting FLAG bit 12) and V2 with a large +// Z (→ small divide → no IR0 saturation). Bit 12 (IR0 sat) is set ONLY by the +// DQ tail's gte_clamp_ir0 and is excluded from the bit-31 summary, so it +// cleanly isolates whether intermediate vertices wrongly ran the DQ. With the +// fix only V2's DQ runs → bit 12 clear. (Pre-fix: V0/V1 saturate it → 0x1000.) +fn tGteRtptDqGate() u32 { + var g: Vec(u32) = gteAlloc(); + gteSetIdentity(g.ptr); + gteCtrlWrite(g.ptr, 26, 2); // H = 2 + gteCtrlWrite(g.ptr, 27, 0x1000); // DQA = 4096 + gteCtrlWrite(g.ptr, 28, 0); // DQB = 0 + gteDataWrite(g.ptr, 0, 0); gteDataWrite(g.ptr, 1, 16); // V0.z = 16 → div 0x2000, IR0 sat + gteDataWrite(g.ptr, 2, 0); gteDataWrite(g.ptr, 3, 16); // V1.z = 16 + gteDataWrite(g.ptr, 4, 0); gteDataWrite(g.ptr, 5, 0x1000); // V2.z = 4096 → div 32, no sat + gteExec(g.ptr, 0x4A000030); // RTPT, sf=0 + return gteCtrlRead(g.ptr, 31) & 0x1000; // FLAG bit 12 = IR0 saturation (DQ tail only) +} + +// NCDS pushes the final colour via the FLAG-aware gte_clamp_rgb (psxe +// cpu.c:1917-1919), so a channel that saturates >255 sets FLAG bit 21/20/19 +// (R/G/B). Drive only the R far-colour large and IR0 to max so stage-5 +// MAC1 = IR0*ir1f hugely overflows 255 on R alone; bits 19/20/21 masked must +// read 0x200000 (R). The old inline clampU8 set none of these (returned 0). +fn tGteNcdsRgbSat() u32 { + var g: Vec(u32) = gteAlloc(); + gteCtrlWrite(g.ptr, 21, 0x00010000); // RFC large → ir1f saturates high + gteDataWrite(g.ptr, 8, 0x7FFF); // IR0 = max + gteExec(g.ptr, 0x4A000013); // NCDS, sf=0 + return gteCtrlRead(g.ptr, 31) & 0x00380000; // RGB-saturation FLAG bits 19/20/21 +} + +// DQA (control reg 27) is s16 (psxe cpu.c:138 / duckstation gte_types.h:117), +// so the RTPS depth-cue must sign-extend its low 16 bits, not read all 32. With +// DQA=0x00010001 the low half is 1: depth-cue MAC0 = DQB + DQA*div = 1*0x10000, +// IR0 = clamp(0x10000>>12) = 16. The old s32 read used 65537, giving a huge +// MAC0 that saturated IR0 to 0x1000. H=SZ3=4096 ⇒ div = 0x10000. +fn tGteDqaWidth() u32 { + var g: Vec(u32) = gteAlloc(); + gteSetIdentity(g.ptr); + gteCtrlWrite(g.ptr, 26, 0x1000); // H = 4096 + gteCtrlWrite(g.ptr, 27, 0x00010001); // DQA: low16=1 (s16), high bits set + gteCtrlWrite(g.ptr, 28, 0); // DQB = 0 + gteDataWrite(g.ptr, 0, 0); + gteDataWrite(g.ptr, 1, 0x1000); // V0.z = 4096 → SZ3 = 4096 + gteExec(g.ptr, 0x4A000001); // RTPS, sf=0 + return gteDataRead(g.ptr, 8); // IR0 = clamp((DQA_s16 * div) >> 12) = 16 +} + tfn psxGteSqrFlag() { assert(tGteSqrFlag(), 0x81000000); } tfn psxGteDivideFlag() { assert(tGteDivideFlag(), 0x80020000); } tfn psxGteNoFlag() { assert(tGteNoFlag(), 0); } @@ -489,6 +559,12 @@ tfn psxGteIrSext() { assert(tGteIrSext(), 0xFFFFFFFF); } tfn psxGteDivide() { assert(tGteDivide(), 0x8000); } tfn psxGteDivideOvf() { assert(tGteDivideOvf(), 0x1FFFF); } +tfn psxGteIrgbNeg() { assert(tGteIrgbNeg(), 0); } +tfn psxGteIrgbPos() { assert(tGteIrgbPos(), 0x3F); } +tfn psxGteRtptDqGate() { assert(tGteRtptDqGate(), 0); } +tfn psxGteNcdsRgbSat() { assert(tGteNcdsRgbSat(), 0x00200000); } +tfn psxGteDqaWidth() { assert(tGteDqaWidth(), 16); } + // ---------- headless emulator boot tests --------------------------------- // // These tests boot the real BIOS (and optionally a disc image) into a diff --git a/timer.jam b/timer.jam index 86c2c46..951f90a 100644 --- a/timer.jam +++ b/timer.jam @@ -38,17 +38,13 @@ pub const TimerChannel = struct { } }; -// Aggregate timer state for the chip. Four shared blank-line flags -// (`hblank` / `prevHblank` / `vblank` / `prevVblank`) drive sync-mode -// behavior; `channels` carries the three 16-bit timers; `subtick` -// is the sub-cycle accumulator for timer 2's /8 clock divisor. +// Aggregate timer state for the chip. Two shared blank-line flags +// (`hblank` / `vblank`) drive sync-mode behavior; `channels` carries the +// three 16-bit timers. pub const Timer = struct { hblank: u32, - prevHblank: u32, vblank: u32, - prevVblank: u32, channels: [3]TimerChannel, - subtick: u32, // Timer 0 dot-clock ticks per sysclk cycle = (11/7)/hdiv, refreshed from // the GPU display mode each frame (setDotclock). Mirrors psxe // timer_get_dotclock_div. Default = 320-wide (hdiv 8). @@ -56,13 +52,12 @@ pub const Timer = struct { pub fn init() Self { return Self { - hblank: 0, prevHblank: 0, vblank: 0, prevVblank: 0, + hblank: 0, vblank: 0, channels: [ TimerChannel.init(), TimerChannel.init(), TimerChannel.init(), ], - subtick: 0, dotDiv: 11.0 / 7.0 / 8.0, }; } @@ -159,19 +154,22 @@ pub const Timer = struct { pub fn handleIrq(self: mut Self, idx: u32, ic: *mut[] u32, cop0: *mut[] u32) { // psxe timer_handle_irq compares the FLOAT counter directly to the - // target and to 65535.0f. + // target and to 65535.0f with strict `>` (timer.c:236-237). jam had + // used `>=` (toward hardware/duckstation), but that fires the timer + // IRQ one tick earlier than psxe on every exact landing — a per-IRQ + // delivery-timing divergence from the FMV's parity reference. Match psxe. const counter: f32 = self.channels[idx].counter; const target: u32 = self.channels[idx].target; var fireIrq: u32 = 0; - if (counter >= (target as f32)) { + if (counter > (target as f32)) { self.channels[idx].targetReached = 1; if (self.channels[idx].resetTarget != 0) { self.channels[idx].counter = 0.0; } if (self.channels[idx].irqTarget != 0) { fireIrq = 1; } } - if (counter >= 65535.0) { + if (counter > 65535.0) { self.channels[idx].counter = 0.0; self.channels[idx].maxReached = 1; if (self.channels[idx].irqMax != 0) { fireIrq = 1; } @@ -299,7 +297,6 @@ pub const Timer = struct { // Hblank / vblank entry/exit hooks called by the GPU's // frame-pacing. pub fn hblankBegin(self: mut Self, ic: *mut[] u32, cop0: *mut[] u32) { - self.prevHblank = self.hblank; self.hblank = 1; if ((self.channels[1].clkSource & 1) != 0 && @@ -340,7 +337,6 @@ pub const Timer = struct { } pub fn vblankBegin(self: mut Self, ic: *mut[] u32, cop0: *mut[] u32) { - self.prevVblank = self.vblank; self.vblank = 1; if (self.channels[1].syncEnable == 0) { return; }