diff --git a/main.jam b/main.jam index 6baad45..823cc70 100644 --- a/main.jam +++ b/main.jam @@ -41,7 +41,7 @@ const { const { sdlInit, sdlQuit, sdlBlit, sdlPump, sdlDelay, - sdlAudioOpen, sdlAudioClose, sdlAudioPush, sdlAudioQueuedBytes + sdlAudioOpen, sdlAudioClose, sdlAudioRingPush, psone_audio_cb } = import("sdl"); const { Vec } = import("std").collections; const { File, exists } = import("std/fs"); @@ -259,8 +259,12 @@ const SCANLINES_TOTAL: u32 = 263; // frames in a GPU scratch u32 slot (slot 37); `line` lives in slot 36. const FloatBits = union { i: u32, f: f32 }; +// Returns the number of stereo audio samples generated this frame (one +// per 768 CPU cycles ≈ 735–737). The main loop paces the frame off this +// count so the emulator advances at exactly the rate that yields 44.1 kHz +// output — keeping SPU production locked to the host's playback rate. fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, - audioDev: u32) { + audioDev: u32, ring: *mut[] u8) u32 { // The scanline counter and the float GPU-cycle accumulator are // CONTINUOUS across frames, exactly like psxe's free-running gpu->line // / gpu->cycles. Persist them in GPU scratch slots 36/37 (the f32 via @@ -365,17 +369,14 @@ fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, gpuBuf[34] = sample; } } - // End-of-frame: feed the SPU samples we generated to SDL. Cap the - // queue depth at ~4 frames (~12 KB) so a host that emulates faster - // than 60 Hz doesn't grow the queue unboundedly — extra samples are - // dropped (host audio stays smooth, the emulator just doesn't - // accumulate latency). SDL drains the queue at the host's audio - // rate (44.1 kHz). + // End-of-frame: push this frame's samples into the SPSC ring. SDL's + // audio callback (psone_audio_cb) drains it at the host 44.1 kHz + // rate; under/overruns are handled at sample granularity inside the + // ring (silence-fill / bounded drop) rather than dropping whole + // frames. sdlAudioRingPush locks the device so it can't race the + // callback. if (audioDev != 0 && audioLen > 0) { - const queued: u32 = sdlAudioQueuedBytes(audioDev); - if (queued < 12000) { - sdlAudioPush(audioDev, audioBuf.asMutPtr(), audioLen); - } + sdlAudioRingPush(audioDev, ring, audioBuf.asMutPtr(), audioLen); } // NOTE: do NOT commitLoad here — each instruction's own commitLoad // handles the R3000 load-delay slot. @@ -383,6 +384,7 @@ fn runOneFrame(c: mut Cpu, bus: Bus, regs: *mut[] u32, cop0: *mut[] u32, var outB: FloatBits = FloatBits { f: acc }; gpuBuf[37] = outB.i; gpuBuf[35] = spuAcc; + return audioLen / 4; // stereo samples produced this frame } // Dump the full 1024x512 VRAM as a binary PPM (P6 / 24-bit RGB) at @@ -578,19 +580,22 @@ fn main() { print("JAM_PSONE_FULL_VRAM set — showing full VRAM\n"); } var sdl: Sdl = sdlInit(title, sdlW, sdlH); - // Audio device — queue mode. Since the SPU is now advanced CPU- - // synchronously inside runOneFrame (one sample per 768 cycles for - // deterministic envelope state — required for Brave Fencer's intro - // to not loop, see commit 51b9ad9), the audio thread doesn't drive - // the SPU anymore. We open without a callback (cbAddr = 0) and feed - // SDL via SDL_QueueAudio (sdlAudioPush) at end of each frame. SDL - // pulls from its internal queue at the host audio rate. - const cbAddr: u64 = 0 as u64; - var audioDev: u32 = sdlAudioOpen(1024, cbAddr, 0 as u64); + // Audio device — callback + SPSC ring buffer. The SPU is advanced + // CPU-synchronously inside runOneFrame (one sample per 768 cycles + // for deterministic envelope state — required for Brave Fencer's + // intro to not loop, see commit 51b9ad9); those samples are pushed + // into `audioRing`, and SDL's audio callback (psone_audio_cb) drains + // the ring at the host 44.1 kHz rate. This replaces queue mode, + // whose whole-frame drops on pacing jitter were the "robotic" gaps. + // Blob = 16-byte header + 16384-byte data region (see sdl.jam RING_*). + var audioRing: [16400]u8 = [0; 16400]; + var ringPtr: *mut[] u8 = audioRing.asMutPtr(); + var audioDev: u32 = sdlAudioOpen(1024, psone_audio_cb as u64, + ringPtr as u64); if (audioDev == 0) { print("WARN: SDL audio device failed to open\n"); } else { - print("Audio device opened (id={audioDev}, 44.1kHz S16 stereo, queue mode)\n"); + print("Audio device opened (id={audioDev}, 44.1kHz S16 stereo, callback+ring)\n"); } print("SDL2 ready.\n"); // VRAM was already zero-filled when the Bus was allocated; the BIOS @@ -655,10 +660,11 @@ fn main() { lastPoly = curPoly; } } - runOneFrame(c, bus, regs.ptr, cop0.ptr, audioDev); - // Audio: runOneFrame fed the per-frame sample batch to SDL via - // SDL_QueueAudio. SDL drains the queue at the host's 44.1 kHz - // pull rate; no callback path active. + const frameSamples: u32 = runOneFrame(c, bus, regs.ptr, cop0.ptr, + audioDev, ringPtr); + // Audio: runOneFrame pushed the per-frame sample batch into the + // ring; SDL's callback (psone_audio_cb) drains it at the host's + // 44.1 kHz pull rate. var vram: *mut[] u8 = busVramPtr(bus); // Read the current display origin out of the GPU state buffer // (g[63] disp_x, g[64] disp_y — set by GP1 0x05). The BIOS @@ -689,11 +695,22 @@ fn main() { // polling runs at VBlank cadence in the BIOS. padSetButtons(bus.pad.ptr, buttons[0]); - // Pace this frame to ~59.94 Hz. The renderer runs WITHOUT vsync (see + // Pace this frame by the AUDIO it produced: frameSamples / 44100 s. + // The SPU emits one sample per 768 CPU cycles, so this makes the + // emulator advance at exactly 33.8688 MHz — the rate at which + // production matches the host's 44.1 kHz playback, so the audio ring + // never drifts full (the old fixed-59.94 Hz target emitted ~737 + // samples/frame = 44178 Hz, a 0.18% excess that crept the ring to + // overrun every few seconds). The renderer runs WITHOUT vsync (see // sdlInit — vsync would block the present on an unfocused window), so - // this wall-clock limiter is the sole pacer: sleep out the rest of the - // 16.68 ms frame so the loop runs at the PSX rate, not flat-out. - nextFrameMs = nextFrameMs + FRAME_MS; + // this wall-clock limiter is the sole pacer. Fall back to the fixed + // target only if a frame somehow produced no audio (avoids a busy + // spin); a normal frame is ~735–737 samples ≈ 16.7 ms. + var frameMs: f64 = FRAME_MS; + if (frameSamples > 0) { + frameMs = (frameSamples as f64) * 1000.0 / 44100.0; + } + nextFrameMs = nextFrameMs + frameMs; const nowMs: u32 = SDL_GetTicks(); if ((nowMs as f64) < nextFrameMs) { SDL_Delay((nextFrameMs - (nowMs as f64)) as u32); diff --git a/sdl.jam b/sdl.jam index 289c205..28c0307 100644 --- a/sdl.jam +++ b/sdl.jam @@ -78,19 +78,22 @@ extern fn objc_msgSend(recv: u64, sel: u64, a: u64, b: u64) u64; // --- audio --- // -// Callback-mode SDL audio. SDL drives the SPU clock: every ~13 ms -// the audio thread calls back into Jam (psone_audio_cb, defined in -// spu.jam as `pub export fn`) for `samplesPerBuffer` stereo frames. -// Each call advances the SPU by exactly that many samples at 44.1 kHz, -// completely decoupled from the CPU loop's wall-clock pace — same -// architecture as psxe (psx/main.c:11-29). Voice pitch stays correct -// even if the emulator runs at 2× or 0.5× host speed. +// Callback-mode SDL audio with an SPSC ring buffer. The SPU is NOT +// driven by the audio thread (it's clocked CPU-synchronously inside +// runOneFrame, one sample per 768 cycles, so the envelope the game +// polls stays deterministic — Brave Fencer's intro depends on this). +// Instead the CPU loop PRODUCES samples into a ring buffer, and SDL's +// audio callback (psone_audio_cb) CONSUMES from it at the rock-steady +// host 44.1 kHz rate. Under/overruns are handled at sample granularity +// (silence-fill on underrun, bounded drop on overrun) instead of the +// whole-frame drops that queue mode (SDL_QueueAudio) produced — which +// were the source of the "robotic" gaps/clicks. // // `cbAddr` is the address of `psone_audio_cb` (computed in main.jam -// via Jam's fn-as-u64 coercion — the new ptrtoint cast lowering in -// the compiler is what makes this expressible without a C shim). -// `userdata` is the AudioCtx blob holding the bus pointers the -// callback needs to reach SPU state. +// via Jam's fn-as-u64 coercion — the ptrtoint cast lowering in the +// compiler is what makes this expressible without a C shim). +// `userdata` is the ring-buffer blob the callback drains (see the +// RING_* layout below). // // SDL_OpenAudioDevice's first arg is `const char*`. We need to pass // NULL to mean "use the default playback device"; Jam can't express @@ -103,6 +106,12 @@ extern fn SDL_CloseAudioDevice(dev: u32); extern fn SDL_PauseAudioDevice(dev: u32, pause_on: i32); extern fn SDL_QueueAudio(dev: u32, data: *const[] u8, len: u32) i32; extern fn SDL_GetQueuedAudioSize(dev: u32) u32; +// Held by the producer (sdlAudioRingPush) while it touches the ring +// header; SDL holds the SAME lock while the callback runs, so producer +// and consumer are never concurrent. jam/std has no atomics, so this +// is how we get cross-thread safety + the needed memory barriers. +extern fn SDL_LockAudioDevice(dev: u32); +extern fn SDL_UnlockAudioDevice(dev: u32); const SDL_INIT_AUDIO: u32 = 0x00000010; // AUDIO_S16SYS on little-endian hosts (macOS) is 0x8010 — signed, @@ -271,16 +280,109 @@ pub fn sdlAudioClose(dev: u32) { if (dev != 0) { SDL_CloseAudioDevice(dev); } } -// Push raw S16LE stereo samples to the device. Each entry is 4 bytes -// (i16 L + i16 R). Returns 0 on success. -pub fn sdlAudioPush(dev: u32, data: *mut[] u8, byteLen: u32) i32 { - if (dev == 0) { return 0; } - return SDL_QueueAudio(dev, data as *const[] u8, byteLen); +// ---- SPSC audio ring buffer --------------------------------------- +// The CPU loop (producer) pushes the per-frame sample batch; SDL's +// audio callback (consumer) drains it at the host rate. Layout of the +// `userdata` blob handed to SDL (indices are byte offsets into the +// CAP-byte data region, CAP a power of two so wrap == & MASK): +// @0 wr u32 producer-owned write offset [0, CAP) +// @4 rd u32 consumer-owned read offset [0, CAP) +// @8 primed u32 0 until PRIME bytes buffered; gates draining so +// we don't thrash on startup / after an underrun +// @16 data [CAP]u8 the ring storage +// Total blob = RING_DATA_OFF + CAP bytes (see main.jam's audioRing). +const RING_CAP: u32 = 16384; // 4096 stereo samples ≈ 93 ms +const RING_MASK: u32 = 16383; // CAP - 1 (CAP is 2^14) +const RING_DATA_OFF: u32 = 16; // header bytes before the data +const RING_PRIME: u32 = 8192; // prebuffer ≈ 46 ms before play + +// Little-endian u32 accessors for the ring header. Byte-wise so there +// are no alignment assumptions on the blob. +pub fn getU32Le(buf: *mut[] u8, off: u32) u32 { + return (buf[off] as u32) + | ((buf[off + 1] as u32) << 8) + | ((buf[off + 2] as u32) << 16) + | ((buf[off + 3] as u32) << 24); +} +pub fn setU32Le(buf: *mut[] u8, off: u32, v: u32) { + buf[off] = (v & 0xFF) as u8; + buf[off + 1] = ((v >> 8) & 0xFF) as u8; + buf[off + 2] = ((v >> 16) & 0xFF) as u8; + buf[off + 3] = ((v >> 24) & 0xFF) as u8; } -pub fn sdlAudioQueuedBytes(dev: u32) u32 { - if (dev == 0) { return 0; } - return SDL_GetQueuedAudioSize(dev); +// PRODUCER (CPU thread): append `n` bytes of S16LE stereo samples from +// `src` into the ring. Locks the device so it can't race the callback. +// On overrun (ring nearly full — rare, since the audio-paced frame +// limiter keeps production matched to the host rate) the OLDEST whole +// frames are dropped to make room, keeping wr/rd frame-aligned. +pub fn sdlAudioRingPush(dev: u32, ring: *mut[] u8, src: *mut[] u8, n: u32) { + if (dev == 0) { return; } + SDL_LockAudioDevice(dev); + var wr: u32 = getU32Le(ring, 0); + var rd: u32 = getU32Le(ring, 4); + // `n` is always a whole number of stereo frames (4 bytes each). EVERY + // index movement here MUST stay a multiple of 4, or wr/rd drift off the + // L/R grid and the callback reads half-frames — which swaps/tears the + // channels into loud noise. (The old code clamped toWrite to `free`, + // but free = RING_CAP-1-avail is ODD, so an overrun wrote an odd byte + // count and permanently misaligned the ring — the "weird audio" bug.) + const avail: u32 = (wr - rd) & RING_MASK; + const free: u32 = RING_CAP - 1 - avail; // 1-byte gap = empty/full + if (n > free) { + // Overrun: jam emits a hair above 44.1 kHz (≈+300 B/s), so the ring + // slowly fills. Make room by dropping the OLDEST whole frames — this + // keeps latency bounded and plays the freshest audio, with at worst a + // faint periodic skip instead of channel-tearing. Safe to move rd + // here: SDL holds the device lock, so the callback can't be running. + var need: u32 = n - free; + need = (need + 3) & 0xFFFFFFFC; // round up to whole frames + rd = (rd + need) & RING_MASK; + setU32Le(ring, 4, rd); + } + // n now fits and is frame-aligned; write the whole batch. + var i: u32 = 0; + while (i < n) { + ring[RING_DATA_OFF + wr] = src[i]; + wr = (wr + 1) & RING_MASK; + i = i + 1; + } + setU32Le(ring, 0, wr); + // Once enough is buffered, allow the callback to start draining. + const after: u32 = (wr - rd) & RING_MASK; + if (after >= RING_PRIME) { setU32Le(ring, 8, 1); } + SDL_UnlockAudioDevice(dev); +} + +// CONSUMER (SDL audio thread): SDL calls this with the device lock +// already held, so it's serialized against sdlAudioRingPush. Drains up +// to `len` bytes from the ring into `stream`; zero-fills any shortfall. +pub export fn psone_audio_cb(userdata: u64, stream: *mut[] u8, len: i32) { + var ring: *mut[] u8 = userdata as *mut[] u8; + const want: u32 = len as u32; + var i: u32 = 0; + // Not yet primed (startup or post-underrun rebuffer) → silence. + if (getU32Le(ring, 8) == 0) { + while (i < want) { stream[i] = 0; i = i + 1; } + return; + } + const wr: u32 = getU32Le(ring, 0); + var rd: u32 = getU32Le(ring, 4); + const avail: u32 = (wr - rd) & RING_MASK; + var toRead: u32 = want; + if (toRead > avail) { toRead = avail; } + while (i < toRead) { + stream[i] = ring[RING_DATA_OFF + rd]; + rd = (rd + 1) & RING_MASK; + i = i + 1; + } + setU32Le(ring, 4, rd); + // Underrun: zero-fill the rest and re-arm priming so we rebuffer + // ~PRIME bytes before resuming (avoids repeated underrun thrash). + if (toRead < want) { + while (i < want) { stream[i] = 0; i = i + 1; } + setU32Le(ring, 8, 0); + } } // Pack an SDL_Rect (4 × i32 = 16 bytes) into a heap buffer. PSX diff --git a/spu.jam b/spu.jam index 6ca2e53..7ca24a6 100644 --- a/spu.jam +++ b/spu.jam @@ -1247,35 +1247,8 @@ pub fn spuGaussInit(s: *mut[] u8) { spuGaussRow(s, 504, 0x5997, 0x599E, 0x59A4, 0x59A9, 0x59AD, 0x59B0, 0x59B2, 0x59B3); } -// ---------- SDL audio callback ---------------------------------------- -// -// Runs on SDL's audio thread at exactly 44.1 kHz. Pulls `len` bytes -// (4 bytes per stereo frame: i16 L + i16 R) by calling spuGetSample -// for each output frame. Sample timing is now owned by the audio -// device — completely decoupled from the CPU loop's wall-clock pace. -// This mirrors psxe (main.c:11-29 / spu.c:575): SPU advances one -// sample per callback iteration. -// -// `userdata` is the address of a 32-byte AudioCtx blob laid out by -// main.jam (offsets 0/8/16/24 = spu / spuram / ic / cop0). The host -// pointer round-trips through u64 — the new IntToPtr cast in the Jam -// compiler is what makes that possible without a C shim. -// -// Thread safety: the audio thread reads SPU register state that the -// CPU thread mutates via spuWrite16/32. Aligned 16/32-bit loads/stores -// are atomic on AArch64, so individual fields don't tear, but voice -// state can briefly desynchronise on KON/KOFF. psxe accepts the same -// race — the audible effect is a one-block (~0.6 ms) glitch in the -// worst case. -pub export fn psone_audio_cb(userdata: u64, stream: *mut[] u8, len: i32) { - // The SPU is now advanced CPU-synchronously from runOneFrame (one - // sample per 768 CPU cycles) so the state the game polls is - // deterministic. The audio thread must NOT advance it (that would - // double-step). Output silence for now; reconnecting audio means - // feeding the CPU-generated samples here via a ring buffer (TODO). - var i: i32 = 0; - while (i < len) { - stream[i] = 0; - i = i + 1; - } -} +// NOTE: the SDL audio callback (psone_audio_cb) lives in sdl.jam now. +// The SPU is clocked CPU-synchronously from runOneFrame (one sample +// per 768 cycles) for deterministic envelope state; those samples are +// pushed into an SPSC ring buffer that the callback drains. The SPU +// has no audio-thread coupling anymore.