From ab8f6e7cb989e40e50764d027bfe5985cd584f9b Mon Sep 17 00:00:00 2001 From: Alex Bates Date: Mon, 23 Feb 2026 21:02:18 +0000 Subject: [PATCH] improve render performance greatly --- CONTRIBUTING.md | 2 +- Cargo.lock | 1 + Cargo.toml | 9 +- crates/kammy/Cargo.toml | 1 + crates/kammy/src/app.rs | 20 +-- crates/kammy/src/dock.rs | 14 +- crates/kammy/src/editor.rs | 1 - crates/kammy/src/editor/display_list.rs | 121 -------------- crates/kammy/src/editor/map.rs | 44 +++--- crates/kammy/src/gpu.rs | 24 +-- crates/kammy/src/main.rs | 20 ++- crates/kammy/src/tool.rs | 7 +- crates/kammy/src/widget/rdp_viewport.rs | 199 ++++++++++++++++-------- crates/parallel_rdp/src/bridge.cpp | 19 ++- crates/parallel_rdp/src/bridge.hpp | 9 ++ crates/parallel_rdp/src/lib.rs | 26 ++++ crates/pm64/Cargo.toml | 1 + crates/pm64/src/gbi.rs | 10 +- crates/pm64/src/render.rs | 141 +++++++++++++++-- crates/rsp/src/lib.rs | 86 +++++++++- crates/rsp/src/rdp.rs | 68 ++++++-- 21 files changed, 527 insertions(+), 296 deletions(-) delete mode 100644 crates/kammy/src/editor/display_list.rs diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 719b5c2..4b1d383 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -34,7 +34,7 @@ If you contribute to an existing file, you may add your own - Write unit tests for all public functions containing business logic, data transformations, or state management. They go in a `#[cfg(test)] mod tests` at the bottom of each file. GUI/rendering code does not require unit tests. - Write integration tests using `egui_kittest` for GUI code. Binary crates (like `kammy`) cannot use the `tests/` directory at the crate root because there is no library target to import. Instead, use a `src/tests.rs` module gated behind `#[cfg(test)]`, with submodules in `src/tests/` organized by feature (e.g. `src/tests/undo.rs`). Shared test utilities go in `src/tests.rs`. Library crates should use the standard `tests/` directory at the crate root. - Prefer module-level inner doc comments (`//!`) at the top of a file over outer doc comments (`///`) on the `mod` declaration. This keeps the documentation next to the code it describes. -- Avoid just `#[expect]` or `#[allow]`ing lines. The checks are there for a reason. For example, `as` should usually be `.into()`. +- Avoid just `#[expect]` or `#[allow]`ing lines. The checks are there for a reason. For example, `as` should usually be `.into()`, or `.try_into()?`. ## Error handling diff --git a/Cargo.lock b/Cargo.lock index 0227e8d..2db86ea 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3128,6 +3128,7 @@ version = "0.1.0" dependencies = [ "loro", "loroscope", + "parallel_rdp", "rsp", ] diff --git a/Cargo.toml b/Cargo.toml index da88954..4f43da2 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -36,9 +36,13 @@ style = { level = "warn", priority = -1 } suspicious = { level = "warn", priority = -1 } perf = { level = "warn", priority = -1 } -must_use_candidate = "allow" redundant_pub_crate = "deny" +# Allows +many_single_char_names = "allow" +must_use_candidate = "allow" +too_many_lines = "allow" + # Don't panic unwrap_used = "deny" expect_used = "warn" @@ -99,3 +103,6 @@ single_component_path_imports = "deny" use_self = "deny" needless_collect = "warn" branches_sharing_code = "warn" + +[profile.dev.package.rsp] +opt-level = 2 diff --git a/crates/kammy/Cargo.toml b/crates/kammy/Cargo.toml index 6aadbb4..2374835 100644 --- a/crates/kammy/Cargo.toml +++ b/crates/kammy/Cargo.toml @@ -10,6 +10,7 @@ edition = "2024" [dependencies] parallel_rdp = { path = "../parallel_rdp" } pm64 = { path = "../pm64" } + loroscope = { path = "../loroscope" } loro = { workspace = true } winit = { workspace = true } diff --git a/crates/kammy/src/app.rs b/crates/kammy/src/app.rs index 3795c86..2f97072 100644 --- a/crates/kammy/src/app.rs +++ b/crates/kammy/src/app.rs @@ -9,7 +9,6 @@ use std::collections::HashMap; use crate::Project; use crate::dock::{Dock, DockPosition}; -use crate::editor::display_list::DisplayListEditor; use crate::editor::map::MapEditor; use crate::editor::todo::TodoEditor; use crate::editor::{Editor, EditorId, Inspect, TileBehavior, UndoBehavior}; @@ -19,6 +18,9 @@ use crate::tool::assets::AssetsTool; use crate::tool::hierarchy::HierarchyTool; use crate::tool::inspector::InspectorTool; +/// Callback for initialising CRDT data when a new editor is added. +type SetupCrdt = dyn Fn(EditorId, &Project); + /// The main application, managing a tabbed editor tree with per-tab undo /// and collapsible tool docks. #[derive(Debug)] @@ -135,7 +137,7 @@ impl KammyApp { fn add_editor( &mut self, make_editor: impl FnOnce(EditorId) -> Box, - setup_crdt: Option<&dyn Fn(EditorId, &Project)>, + setup_crdt: Option<&SetupCrdt>, ) { let editor_id = self.alloc_editor_id(); @@ -188,10 +190,6 @@ impl KammyApp { ); } - fn add_display_list_editor(&mut self) { - self.add_editor(|id| Box::new(DisplayListEditor::new(id)), None); - } - fn add_map_editor(&mut self) { self.add_editor( |id| Box::new(MapEditor::new(id)), @@ -345,9 +343,6 @@ impl KammyApp { if ui.button("+ Todo").clicked() { self.add_todo_editor(); } - if ui.button("+ Display List").clicked() { - self.add_display_list_editor(); - } if ui.button("+ Map").clicked() { self.add_map_editor(); } @@ -394,7 +389,6 @@ impl KammyApp { // Destructure for disjoint borrows let Self { project, - active_editor_id, inspect, left_dock, right_dock, @@ -402,11 +396,7 @@ impl KammyApp { .. } = self; - let mut tool_ctx = ToolContext { - project, - active_editor_id: *active_editor_id, - inspect, - }; + let mut tool_ctx = ToolContext { inspect }; bottom_dock.show(ctx, &mut tool_ctx); left_dock.show(ctx, &mut tool_ctx); diff --git a/crates/kammy/src/dock.rs b/crates/kammy/src/dock.rs index c3cf2e1..7d83901 100644 --- a/crates/kammy/src/dock.rs +++ b/crates/kammy/src/dock.rs @@ -52,11 +52,6 @@ impl Dock { } } - /// Whether the dock is currently open (has an active tool). - pub fn is_open(&self) -> bool { - self.active.is_some() - } - /// Toggles a tool by index. If the tool is already active, collapses the /// dock. Otherwise, activates the tool. pub fn toggle_tool(&mut self, idx: usize) { @@ -130,8 +125,6 @@ impl Dock { mod tests { use super::*; - use egui; - #[derive(Debug)] struct DummyTool { title: &'static str, @@ -169,13 +162,13 @@ mod tests { #[test] fn starts_collapsed() { let dock = make_dock(None); - assert!(!dock.is_open()); + assert_eq!(dock.active, None); } #[test] fn starts_open() { let dock = make_dock(Some(0)); - assert!(dock.is_open()); + assert_eq!(dock.active, Some(0)); } #[test] @@ -184,12 +177,10 @@ mod tests { dock.toggle_tool(0); assert_eq!(dock.active, Some(0)); - assert!(dock.is_open()); // Toggle same tool collapses dock.toggle_tool(0); assert_eq!(dock.active, None); - assert!(!dock.is_open()); } #[test] @@ -198,6 +189,5 @@ mod tests { dock.toggle_tool(1); assert_eq!(dock.active, Some(1)); - assert!(dock.is_open()); } } diff --git a/crates/kammy/src/editor.rs b/crates/kammy/src/editor.rs index 078fb44..1a16593 100644 --- a/crates/kammy/src/editor.rs +++ b/crates/kammy/src/editor.rs @@ -4,7 +4,6 @@ //! Editor trait, built-in editor implementations, and tile-tree dispatch. -pub mod display_list; pub mod map; pub mod todo; diff --git a/crates/kammy/src/editor/display_list.rs b/crates/kammy/src/editor/display_list.rs deleted file mode 100644 index e28cd5b..0000000 --- a/crates/kammy/src/editor/display_list.rs +++ /dev/null @@ -1,121 +0,0 @@ -// SPDX-FileCopyrightText: 2026 Alex Bates -// -// SPDX-License-Identifier: AGPL-3.0-or-later - -//! Display list editor: renders N64 display lists via parallel-rdp. -//! -//! Currently renders a solid-color test pattern to verify the full pipeline: -//! RDRAM write -> RDP command submit -> scanout -> wgpu texture -> egui display. - -use super::{Editor, EditorContext, EditorId}; -use crate::widget::rdp_viewport::{DisplayList, RdpViewport, ViConfig}; - -/// An editor that renders N64 display lists via the RDP. -pub struct DisplayListEditor { - id: EditorId, - viewport: RdpViewport, - frame_count: u32, -} - -impl std::fmt::Debug for DisplayListEditor { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("DisplayListEditor") - .field("id", &self.id) - .field("frame_count", &self.frame_count) - .finish_non_exhaustive() - } -} - -impl DisplayListEditor { - /// Creates a new display list editor with the given stable ID. - pub fn new(id: EditorId) -> Self { - Self { - id, - viewport: RdpViewport::new(4 * 1024 * 1024), - frame_count: 0, - } - } -} - -/// Build an RDP display list that fills the framebuffer with a solid color. -/// -/// The color cycles through red, green, blue based on the frame counter, -/// producing a simple animated test pattern. -fn build_fill_rect_display_list(frame: u32) -> DisplayList { - // Framebuffer: 320x240, 16-bit (5/5/5/1) - const FB_WIDTH: u32 = 320; - const FB_HEIGHT: u32 = 240; - const FB_ORIGIN: u32 = 0x100; // Non-zero: parallel-rdp treats origin 0 as blank - - let phase = frame / 60 % 3; - let fill_color: u32 = match phase { - 0 => 0xF801_F801, // Red (16-bit 5551: R=31, G=0, B=0, A=1), packed twice - 1 => 0x07C1_07C1, // Green - _ => 0x003F_003F, // Blue - }; - - // RDP commands (each command is 64 bits = 2 words) - let commands: Vec = vec![ - // Set Color Image: format=RGBA, size=16-bit, width=320, address=0 - // Command byte: 0x3F (Set Color Image) - // Bits: [63:56]=0x3F, [55:53]=format(0=RGBA), [52:51]=size(1=16-bit), - // [41:32]=width-1, [25:0]=address - 0x3F10_0000 | ((FB_WIDTH - 1) & 0x3FF), - FB_ORIGIN, - // Set Scissor: XH=0, YH=0, XL=320<<2, YL=240<<2 - // Command byte: 0x2D - 0x2D00_0000, - ((FB_WIDTH << 2) << 12) | (FB_HEIGHT << 2), - // Set Other Modes: cycle_type=Fill - // Command byte: 0x2F, bit 55-52 = cycle type (3=Fill) - 0x2F30_0000, - 0x0000_0000, - // Set Fill Color - // Command byte: 0x37 - 0x3700_0000, - fill_color, - // Fill Rectangle: covers entire framebuffer - // Command byte: 0x36 - // Bits: XL=320<<2, YL=240<<2 (word 0), XH=0, YH=0 (word 1) - 0x3600_0000 | ((FB_WIDTH << 2) << 12) | (FB_HEIGHT << 2), - 0x0000_0000, - // Sync Full: wait for all rendering to complete - // Command byte: 0x29 - 0x2900_0000, - 0x0000_0000, - ]; - - let vi = ViConfig { - // Control: 16-bit color (bits 1:0 = 2), anti-alias + resample (bits 9:8 = 3) - control: 0x0000_0302, - origin: FB_ORIGIN, - width: FB_WIDTH, - v_sync: 525, // NTSC: 525 lines - h_start: (0x006C << 16) | 0x02EC, // Typical NTSC H range - v_start: (0x0025 << 16) | 0x01FF, // Typical NTSC V range - x_scale: (FB_WIDTH * 1024 / 640), // Scale to fill 640 output - y_scale: (FB_HEIGHT * 1024 / 480), // Scale to fill 480 output - }; - - DisplayList { commands, vi } -} - -impl Editor for DisplayListEditor { - fn id(&self) -> EditorId { - self.id - } - - fn title(&self) -> String { - "Display List".to_owned() - } - - fn ui(&mut self, ui: &mut egui::Ui, ctx: &mut EditorContext) { - let display_list = build_fill_rect_display_list(self.frame_count); - self.frame_count = self.frame_count.wrapping_add(1); - - self.viewport - .ui(ui, ctx.gpu.as_deref_mut(), &display_list, 4.0 / 3.0, |_| {}); - - ui.ctx().request_repaint(); - } -} diff --git a/crates/kammy/src/editor/map.rs b/crates/kammy/src/editor/map.rs index 7b117a9..8b31010 100644 --- a/crates/kammy/src/editor/map.rs +++ b/crates/kammy/src/editor/map.rs @@ -15,7 +15,7 @@ use pm64::model::ModelNode; use super::{Editor, EditorContext, EditorId}; use crate::Project; -use crate::widget::rdp_viewport::{DisplayList, RdpViewport}; +use crate::widget::rdp_viewport::RdpViewport; const FB_WIDTH: u32 = 320; const FB_HEIGHT: u32 = 240; @@ -27,8 +27,6 @@ pub struct MapEditor { id: EditorId, viewport: RdpViewport, camera: camera::OrbitCamera, - /// Persistent RSP renderer — avoids 4 MB RDRAM allocation per frame. - rsp_renderer: pm64::render::Renderer, } impl std::fmt::Debug for MapEditor { @@ -46,7 +44,6 @@ impl MapEditor { id, viewport: RdpViewport::new(4 * 1024 * 1024), camera: camera::OrbitCamera::default(), - rsp_renderer: pm64::render::Renderer::new(), } } } @@ -61,32 +58,33 @@ impl Editor for MapEditor { } fn ui(&mut self, ui: &mut egui::Ui, ctx: &mut EditorContext) { + // Handle camera input first so this frame's drag is reflected immediately. + let interact_rect = ui.available_rect_before_wrap(); + let interact_response = ui.interact( + interact_rect, + ui.id().with("camera"), + egui::Sense::click_and_drag(), + ); + self.camera.handle_input(&interact_response); + let nodes = extract_nodes(ctx.project); - let aspect = FB_WIDTH as f32 / FB_HEIGHT as f32; + #[expect( + clippy::cast_possible_truncation, + clippy::as_conversions, + reason = "320/240 is well within f32 range" + )] + let aspect = (f64::from(FB_WIDTH) / f64::from(FB_HEIGHT)) as f32; let camera_matrices = self.camera.to_n64_matrices(aspect); + let vi = vi_config_ntsc(FB_ORIGIN); - let rdp_commands = self.rsp_renderer.render(&nodes, &camera_matrices); - - let display_list = DisplayList { - commands: rdp_commands, - vi: vi_config_ntsc(FB_ORIGIN), - }; - - let response = self.viewport.ui( + self.viewport.ui( ui, ctx.gpu.as_deref_mut(), - &display_list, + &vi, aspect, - |_rdram| {}, - ); - - // Layer a drag/scroll sensor over the viewport for camera control - let response = ui.interact( - response.rect, - response.id.with("camera"), - egui::Sense::click_and_drag(), + &nodes, + &camera_matrices, ); - self.camera.handle_input(&response); ui.ctx().request_repaint(); } diff --git a/crates/kammy/src/gpu.rs b/crates/kammy/src/gpu.rs index b146920..49ca306 100644 --- a/crates/kammy/src/gpu.rs +++ b/crates/kammy/src/gpu.rs @@ -22,21 +22,25 @@ use raw_window_handle::HasDisplayHandle; use winit::window::Window; /// GPU state shared across the application. +/// +/// Field order matters for drop: wgpu resources must be released before +/// `rdp_context` destroys the underlying VkDevice/VkInstance. pub struct GpuState { - /// The parallel-rdp Vulkan context (owns the `VkInstance` + `VkDevice`). - pub rdp_context: parallel_rdp::VulkanContext, - /// wgpu device wrapping Granite's `VkDevice`. - pub device: wgpu::Device, - /// wgpu queue wrapping Granite's graphics queue. - pub queue: wgpu::Queue, - /// Window surface for presentation. - surface: wgpu::Surface<'static>, - /// Current surface configuration. - surface_config: wgpu::SurfaceConfiguration, /// egui renderer (draws egui primitives via wgpu). pub renderer: egui_wgpu::Renderer, + /// Current surface configuration. + surface_config: wgpu::SurfaceConfiguration, + /// Window surface for presentation. + surface: wgpu::Surface<'static>, + /// wgpu queue wrapping Granite's graphics queue. + pub queue: wgpu::Queue, + /// wgpu device wrapping Granite's `VkDevice`. + pub device: wgpu::Device, /// Prevents the wgpu Instance from being dropped prematurely. _instance: wgpu::Instance, + /// The parallel-rdp Vulkan context (owns the `VkInstance` + `VkDevice`). + /// Must be last: Granite owns the Vulkan handles that everything above wraps. + pub rdp_context: parallel_rdp::VulkanContext, } impl std::fmt::Debug for GpuState { diff --git a/crates/kammy/src/main.rs b/crates/kammy/src/main.rs index c9d70af..cf9c0c8 100644 --- a/crates/kammy/src/main.rs +++ b/crates/kammy/src/main.rs @@ -55,12 +55,16 @@ struct WinitApp { } /// Runtime state created after the window is available. +/// +/// Field order matters: Rust drops fields in declaration order, and the +/// app's editors hold parallel-rdp `Renderer`s whose destructors need +/// the Vulkan device that lives inside `gpu`. So `app` must drop first. struct AppState { - window: Arc, - gpu: gpu::GpuState, - egui_ctx: egui::Context, - egui_state: egui_winit::State, app: app::KammyApp, + egui_state: egui_winit::State, + egui_ctx: egui::Context, + gpu: gpu::GpuState, + window: Arc, } impl ApplicationHandler for WinitApp { @@ -106,11 +110,11 @@ impl ApplicationHandler for WinitApp { let app = app::KammyApp::new(); self.state = Some(AppState { - window, - gpu, - egui_ctx, - egui_state, app, + egui_state, + egui_ctx, + gpu, + window, }); } diff --git a/crates/kammy/src/tool.rs b/crates/kammy/src/tool.rs index 1611364..34b3e0a 100644 --- a/crates/kammy/src/tool.rs +++ b/crates/kammy/src/tool.rs @@ -12,15 +12,10 @@ pub mod assets; pub mod hierarchy; pub mod inspector; -use crate::Project; -use crate::editor::{EditorId, Inspect}; +use crate::editor::Inspect; /// Context passed to each tool during rendering. pub struct ToolContext<'a> { - /// The shared project data (CRDT document). - pub project: &'a Project, - /// The currently focused editor, if any. - pub active_editor_id: Option, /// The current inspect object set by editors. Tools like the Inspector /// read this to display property UI. pub inspect: &'a mut Option>, diff --git a/crates/kammy/src/widget/rdp_viewport.rs b/crates/kammy/src/widget/rdp_viewport.rs index 60e1cbf..05efe65 100644 --- a/crates/kammy/src/widget/rdp_viewport.rs +++ b/crates/kammy/src/widget/rdp_viewport.rs @@ -6,6 +6,9 @@ //! An egui widget that renders N64 display lists using parallel-rdp. +use pm64::gbi::{CameraMatrices, NodeData}; +use pm64::render::ParallelRdpSink; + use crate::gpu::GpuState; /// N64 Video Interface register configuration for scanout. @@ -29,31 +32,40 @@ pub struct ViConfig { pub y_scale: u32, } -/// A display list to be rendered by the RDP. -#[derive(Debug, Clone)] -pub struct DisplayList { - /// RDP command words (big-endian 32-bit). - pub commands: Vec, - /// Video Interface configuration for scanout. - pub vi: ViConfig, -} - /// Reusable egui widget that renders N64 display lists via parallel-rdp. /// /// Each instance owns its own [`parallel_rdp::Renderer`] (command processor + /// RDRAM). The widget submits display list commands, performs scanout, and /// displays the result as an egui image. /// -/// The renderer is created lazily on the first [`show`](Self::show) call that +/// GPU work is pipelined: each frame submits commands and signals the GPU +/// timeline (non-blocking), then waits for the *previous* frame's signal +/// at the start of the next frame. This overlaps GPU rendering with the +/// CPU's egui layout pass, eliminating the blocking `flush()` stall. +/// +/// The renderer is created lazily on the first [`ui`](Self::ui) call that /// receives a GPU context. pub struct RdpViewport { renderer: Option, rdram_size: u32, + /// Persistent RSP renderer — avoids 4 MB RDRAM allocation per frame. + rsp_renderer: pm64::render::Renderer, /// Registered egui texture ID (reused across frames). texture_id: Option, /// The current frame's scanout texture wrapper. Kept alive so egui can - /// reference it during the render pass (which runs after `show()`). + /// reference it during the render pass (which runs after `ui()`). current_texture: Option, + /// Pending GPU timeline value from the previous frame's scanout. + pending_timeline: Option, + /// Scanout result waiting for the timeline to complete. + pending_scanout: Option, +} + +/// A scanout result waiting to be imported into wgpu once the GPU finishes. +struct PendingScanout { + vk_image: ash::vk::Image, + width: u32, + height: u32, } impl std::fmt::Debug for RdpViewport { @@ -64,109 +76,171 @@ impl std::fmt::Debug for RdpViewport { } } +impl Drop for RdpViewport { + fn drop(&mut self) { + // Wait for any in-flight GPU work before destroying the renderer. + if let (Some(timeline), Some(renderer)) = + (self.pending_timeline.take(), self.renderer.as_mut()) + { + renderer.wait_for_timeline(timeline); + } + // Release the wgpu texture wrapping a VkImage owned by the renderer + // before the renderer (and its CommandProcessor) are dropped. + self.current_texture = None; + } +} + impl RdpViewport { /// Creates a new viewport. /// /// `rdram_size` is the RDRAM capacity in bytes (typically 4 MiB). The - /// underlying renderer is created lazily when [`show`](Self::show) is + /// underlying renderer is created lazily when [`ui`](Self::ui) is /// first called with a GPU context. pub fn new(rdram_size: u32) -> Self { Self { renderer: None, rdram_size, + rsp_renderer: pm64::render::Renderer::new(), texture_id: None, current_texture: None, + pending_timeline: None, + pending_scanout: None, } } - /// Renders the display list and shows the result in the UI. + /// Renders an N64 frame and shows the result in the UI. /// /// `display_aspect` is the intended display aspect ratio (width/height). /// The scanout texture is stretched to fill the available UI space at /// this ratio — necessary because non-interlaced VI modes produce /// half-height scanouts that don't reflect the true display shape. /// - /// The closure receives the renderer's RDRAM for direct writes (textures, - /// framebuffer data, etc.) before commands are submitted. + /// Has 1 frame of latency. /// /// If `gpu` is `None` (headless/test), displays a placeholder label. pub fn ui( &mut self, ui: &mut egui::Ui, gpu: Option<&mut GpuState>, - display_list: &DisplayList, + vi: &ViConfig, display_aspect: f32, - write_rdram: impl FnOnce(&mut [u8]), + nodes: &[NodeData], + camera: &CameraMatrices, ) -> egui::Response { let Some(gpu) = gpu else { return ui.label("GPU not available"); }; - let renderer = match &mut self.renderer { - Some(r) => r, - None => match parallel_rdp::Renderer::new(&gpu.rdp_context, self.rdram_size, 0) { - Ok(r) => self.renderer.insert(r), + // Lazily create the renderer on first use. + if self.renderer.is_none() { + match parallel_rdp::Renderer::new(&gpu.rdp_context, self.rdram_size, 0) { + Ok(r) => { + self.renderer = Some(r); + } Err(e) => { tracing::warn!("failed to create RDP renderer: {e:?}"); return ui.label("RDP renderer unavailable"); } - }, - }; + } + } - write_rdram(renderer.rdram_mut()); - renderer.begin_frame(); - Self::set_vi_registers(renderer, &display_list.vi); - renderer.enqueue_commands(&display_list.commands); + // Wait for the previous frame's GPU work and import its scanout. + // This wait should be near-instant because the GPU has had a full + // egui frame (~16ms) to finish since we signalled. + // + // Scoped separately from the command submission below so + // `update_egui_texture` can borrow `&mut self`. + if let Some(timeline) = self.pending_timeline.take() { + if let Some(renderer) = &mut self.renderer { + renderer.wait_for_timeline(timeline); + } - let Some((vk_image, width, height)) = renderer.scanout() else { - return ui.label("No scanout output"); - }; - if width == 0 || height == 0 { - return ui.label("No scanout output"); + if let Some(scanout) = self.pending_scanout.take() { + // SAFETY: wait_for_timeline ensures the GPU is done, and the + // VkImage is still valid (no new scanout has been called yet). + if let Some(texture) = unsafe { + import_scanout_image( + &gpu.device, + scanout.vk_image, + scanout.width, + scanout.height, + ) + } { + let view = texture.create_view(&wgpu::TextureViewDescriptor::default()); + self.update_egui_texture(gpu, &view); + self.current_texture = Some(texture); + } else { + tracing::warn!("failed to import scanout VkImage into wgpu"); + } + } } - // Ensure all GPU scanout work is complete before wgpu reads the image - renderer.flush(); + // Submit new work for this frame. + let Some(renderer) = &mut self.renderer else { + return ui.label("RDP renderer unavailable"); + }; + renderer.begin_frame(); + Self::set_vi_registers(renderer, vi); + self.rsp_renderer + .render_to(nodes, camera, &mut ParallelRdpSink(renderer)); - // SAFETY: flush() was called above, and the VkImage from scanout() - // remains valid until the wgpu::Texture is dropped (next frame at earliest). - let Some(texture) = (unsafe { import_scanout_image(&gpu.device, vk_image, width, height) }) - else { - tracing::warn!("failed to import scanout VkImage into wgpu"); - return ui.label("Scanout import failed"); + if let Some((vk_image, width, height)) = renderer.scanout() + && width > 0 + && height > 0 + { + if self.current_texture.is_none() { + // First frame: no previous texture to display yet, so + // do a blocking flush to bootstrap. + renderer.flush(); + if let Some(texture) = + unsafe { import_scanout_image(&gpu.device, vk_image, width, height) } + { + let view = texture.create_view(&wgpu::TextureViewDescriptor::default()); + self.update_egui_texture(gpu, &view); + self.current_texture = Some(texture); + } + } else { + // Pipeline: signal non-blocking, import on next frame + self.pending_timeline = Some(renderer.signal_timeline()); + self.pending_scanout = Some(PendingScanout { + vk_image, + width, + height, + }); + } + } + + // Display the texture (either from previous frame's import or + // from the bootstrap flush above) + let available = ui.available_size(); + let size = if available.x / available.y.max(1.0) > display_aspect { + egui::vec2(available.y * display_aspect, available.y) + } else { + egui::vec2(available.x, available.x / display_aspect) }; - let view = texture.create_view(&wgpu::TextureViewDescriptor::default()); - // Register or update the egui texture binding + if let Some(texture_id) = self.texture_id { + ui.image(egui::load::SizedTexture::new(texture_id, size)) + } else { + ui.label("Loading…") + } + } + + /// Registers or updates the egui texture binding. + fn update_egui_texture(&mut self, gpu: &mut GpuState, view: &wgpu::TextureView) { if let Some(id) = self.texture_id { gpu.renderer.update_egui_texture_from_wgpu_texture( &gpu.device, - &view, + view, wgpu::FilterMode::Nearest, id, ); } else { let id = gpu.renderer - .register_native_texture(&gpu.device, &view, wgpu::FilterMode::Nearest); + .register_native_texture(&gpu.device, view, wgpu::FilterMode::Nearest); self.texture_id = Some(id); } - - // Keep texture alive until the render pass uses it - self.current_texture = Some(texture); - - // Scale image to fill available UI space at the caller's display aspect ratio - let available = ui.available_size(); - let size = if available.x / available.y.max(1.0) > display_aspect { - egui::vec2(available.y * display_aspect, available.y) - } else { - egui::vec2(available.x, available.x / display_aspect) - }; - - let Some(texture_id) = self.texture_id else { - return ui.label("Texture not ready"); - }; - ui.image(egui::load::SizedTexture::new(texture_id, size)) } fn set_vi_registers(renderer: &mut parallel_rdp::Renderer, vi: &ViConfig) { @@ -186,8 +260,9 @@ impl RdpViewport { /// /// # Safety /// -/// The `VkImage` must be valid and fully rendered (call `flush()` first). -/// It must remain valid until the wgpu texture is dropped. +/// The `VkImage` must be valid and fully rendered (call `flush()` or +/// `wait_for_timeline()` first). It must remain valid until the wgpu +/// texture is dropped. unsafe fn import_scanout_image( device: &wgpu::Device, vk_image: ash::vk::Image, diff --git a/crates/parallel_rdp/src/bridge.cpp b/crates/parallel_rdp/src/bridge.cpp index 842d88a..d5acde0 100644 --- a/crates/parallel_rdp/src/bridge.cpp +++ b/crates/parallel_rdp/src/bridge.cpp @@ -157,7 +157,12 @@ void *rdp_renderer_create(void *ctx, uint32_t rdram_size, uint32_t flags) void rdp_renderer_destroy(void *renderer) { - delete static_cast(renderer); + auto *r = static_cast(renderer); + // Ensure all GPU work completes before destroying the CommandProcessor, + // otherwise its destructor may race with in-flight commands. + uint64_t timeline = r->processor->signal_timeline(); + r->processor->wait_for_timeline(timeline); + delete r; } uint8_t *rdp_renderer_get_rdram(void *renderer) @@ -283,3 +288,15 @@ void rdp_renderer_flush(void *renderer) uint64_t timeline = r->processor->signal_timeline(); r->processor->wait_for_timeline(timeline); } + +uint64_t rdp_renderer_signal_timeline(void *renderer) +{ + auto *r = static_cast(renderer); + return r->processor->signal_timeline(); +} + +void rdp_renderer_wait_for_timeline(void *renderer, uint64_t value) +{ + auto *r = static_cast(renderer); + r->processor->wait_for_timeline(value); +} diff --git a/crates/parallel_rdp/src/bridge.hpp b/crates/parallel_rdp/src/bridge.hpp index 63c0c72..5cabcd9 100644 --- a/crates/parallel_rdp/src/bridge.hpp +++ b/crates/parallel_rdp/src/bridge.hpp @@ -120,6 +120,15 @@ int rdp_renderer_scanout_sync( /// Signal the renderer's timeline and wait for all previous work to complete. void rdp_renderer_flush(void *renderer); +/// Signal the renderer's timeline and return the timeline value (non-blocking). +/// +/// Call `rdp_renderer_wait_for_timeline` with the returned value to wait +/// for all work submitted before this signal to complete. +uint64_t rdp_renderer_signal_timeline(void *renderer); + +/// Wait for the renderer's timeline to reach `value`. +void rdp_renderer_wait_for_timeline(void *renderer, uint64_t value); + #ifdef __cplusplus } #endif diff --git a/crates/parallel_rdp/src/lib.rs b/crates/parallel_rdp/src/lib.rs index 4f4c412..1b59237 100644 --- a/crates/parallel_rdp/src/lib.rs +++ b/crates/parallel_rdp/src/lib.rs @@ -271,6 +271,15 @@ impl Renderer { /// /// Write display list data, textures, and framebuffer contents here before /// calling [`enqueue_commands`](Self::enqueue_commands) and [`scanout`](Self::scanout). + /// + /// # Panics + /// + /// Panics if the RDRAM size (a `u32`) does not fit in a `usize`. This can + /// only happen on 16-bit platforms, which cannot run Vulkan. + #[expect( + clippy::expect_used, + reason = "RDRAM size is u32, which fits in usize on all Vulkan-capable platforms" + )] pub fn rdram_mut(&mut self) -> &mut [u8] { unsafe { let ptr = ffi::rdp_renderer_get_rdram(self.ptr); @@ -364,6 +373,23 @@ impl Renderer { ffi::rdp_renderer_flush(self.ptr); } } + + /// Signals the GPU timeline and returns a token (non-blocking). + /// + /// Call [`wait_for_timeline`](Self::wait_for_timeline) with the returned + /// value to wait for all work submitted before this signal. + pub fn signal_timeline(&mut self) -> u64 { + unsafe { ffi::rdp_renderer_signal_timeline(self.ptr) } + } + + /// Waits for the GPU timeline to reach the given value. + /// + /// If the GPU has already passed this point, returns immediately. + pub fn wait_for_timeline(&mut self, value: u64) { + unsafe { + ffi::rdp_renderer_wait_for_timeline(self.ptr, value); + } + } } impl Drop for Renderer { diff --git a/crates/pm64/Cargo.toml b/crates/pm64/Cargo.toml index 010bae4..468d07d 100644 --- a/crates/pm64/Cargo.toml +++ b/crates/pm64/Cargo.toml @@ -9,6 +9,7 @@ edition = "2024" [dependencies] loroscope = { path = "../loroscope" } +parallel_rdp = { path = "../parallel_rdp" } rsp = { path = "../rsp" } loro = { workspace = true } diff --git a/crates/pm64/src/gbi.rs b/crates/pm64/src/gbi.rs index d69b67f..e8ac48a 100644 --- a/crates/pm64/src/gbi.rs +++ b/crates/pm64/src/gbi.rs @@ -20,7 +20,7 @@ pub struct VertexData { } /// A triangle referencing three vertices. -#[derive(Clone, Debug)] +#[derive(Clone, Debug, PartialEq)] pub struct TriangleData { pub v0: VertexData, pub v1: VertexData, @@ -28,14 +28,14 @@ pub struct TriangleData { } /// A model node's geometry in plain (non-CRDT) form. -#[derive(Clone, Debug)] +#[derive(Clone, Debug, PartialEq)] pub struct NodeData { /// Triangles belonging to this node. pub triangles: Vec, } /// N64 camera matrices in s15.16 fixed-point format (64 bytes each). -#[derive(Clone, Debug)] +#[derive(Clone, Debug, PartialEq, Eq)] pub struct CameraMatrices { /// Projection matrix (64 bytes, s15.16 fixed-point). pub projection: [u8; 64], @@ -156,10 +156,6 @@ fn pack_viewport(width: u32, height: u32) -> [u8; 16] { /// - `proj_addr`: RDRAM address of the projection matrix. /// - `mv_addr`: RDRAM address of the modelview matrix. /// - `viewport_addr`: RDRAM address where the viewport struct will be placed. -#[expect( - clippy::many_single_char_names, - reason = "a/b/c vertex indices are standard triangle nomenclature" -)] pub fn reconstruct( nodes: &[NodeData], _camera: &CameraMatrices, diff --git a/crates/pm64/src/render.rs b/crates/pm64/src/render.rs index 5895799..943ece8 100644 --- a/crates/pm64/src/render.rs +++ b/crates/pm64/src/render.rs @@ -9,6 +9,16 @@ use crate::gbi::{self, CameraMatrices, GbiOutput, NodeData}; +/// Bridges [`rsp::RdpSink`] to [`parallel_rdp::Renderer::enqueue_commands`]. +#[derive(Debug)] +pub struct ParallelRdpSink<'a>(pub &'a mut parallel_rdp::Renderer); + +impl rsp::RdpSink for ParallelRdpSink<'_> { + fn receive_commands(&mut self, commands: &[u32]) { + self.0.enqueue_commands(commands); + } +} + // Microcode binaries from the Paper Mario 64 decomp const F3DEX2_TEXT: &[u8] = include_bytes!(concat!( env!("PM64_ASSETS_DIR"), @@ -115,7 +125,7 @@ fn write_be_bytes_to_rdram(rdram: &mut [u8], offset: usize, data: &[u8]) { /// on every call. pub fn render(nodes: &[NodeData], camera: &CameraMatrices) -> Vec { let mut renderer = Renderer::new(); - renderer.render(nodes, camera) + renderer.render(nodes, camera).to_vec() } /// Persistent RSP render context that reuses its device across frames. @@ -123,14 +133,24 @@ pub fn render(nodes: &[NodeData], camera: &CameraMatrices) -> Vec { /// Avoids the 4 MB RDRAM allocation that [`render`] incurs on every call. /// Microcode is loaded once at construction; subsequent [`render`](Self::render) /// calls only write the per-frame data (display list, vertices, matrices) -/// and reset the RSP/RDP state. +/// and reset the RSP/RDP control state. pub struct Renderer { device: rsp::Device, + /// `true` after the first frame (IMEM holds F3DEX2, not rspboot). + warm: bool, } impl std::fmt::Debug for Renderer { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("Renderer").finish_non_exhaustive() + f.debug_struct("Renderer") + .field("warm", &self.warm) + .finish_non_exhaustive() + } +} + +impl Default for Renderer { + fn default() -> Self { + Self::new() } } @@ -141,11 +161,17 @@ impl Renderer { let rdram = device.rdram_mut(); write_be_bytes_to_rdram(rdram, F3DEX2_TEXT_ADDR, F3DEX2_TEXT); write_be_bytes_to_rdram(rdram, F3DEX2_DATA_ADDR, F3DEX2_DATA); - Self { device } + Self { + device, + warm: false, + } } /// Renders model geometry through the RSP, producing RDP command words. - pub fn render(&mut self, nodes: &[NodeData], camera: &CameraMatrices) -> Vec { + /// + /// The returned slice borrows from the internal RSP device and is valid + /// until the next call to `render`. + pub fn render(&mut self, nodes: &[NodeData], camera: &CameraMatrices) -> &[u32] { let gbi_output = gbi::reconstruct( nodes, camera, @@ -158,9 +184,45 @@ impl Renderer { self.render_gbi(&gbi_output, camera) } + /// Renders model geometry. + pub fn render_to( + &mut self, + nodes: &[NodeData], + camera: &CameraMatrices, + sink: &mut dyn rsp::RdpSink, + ) { + let gbi_output = gbi::reconstruct( + nodes, + camera, + addr_u32(FB_ADDR), + addr_u32(VERTEX_ADDR), + addr_u32(PROJ_MTX_ADDR), + addr_u32(MV_MTX_ADDR), + addr_u32(VIEWPORT_ADDR), + ); + self.prepare_frame(&gbi_output, camera); + self.device.run_with_sink(sink); + } + /// Renders a pre-built GBI display list through the RSP. - fn render_gbi(&mut self, gbi_output: &GbiOutput, camera: &CameraMatrices) -> Vec { - self.device.reset(); + fn render_gbi(&mut self, gbi_output: &GbiOutput, camera: &CameraMatrices) -> &[u32] { + self.prepare_frame(gbi_output, camera); + self.device.run() + } + + /// Resets the RSP/RDP, writes all per-frame data (display list, vertices, + /// matrices, `OSTask`) to RDRAM/DMEM, and prepares the RSP for execution. + /// + /// On the first frame, a full [`reset`](rsp::Device::reset) is used. On + /// subsequent frames, [`rearm`](rsp::Device::rearm) resets only the + /// control registers while preserving lookup tables and dispatch tables. + /// Rspboot always runs to properly initialise F3DEX2. + fn prepare_frame(&mut self, gbi_output: &GbiOutput, camera: &CameraMatrices) { + if self.warm { + self.device.rearm(); + } else { + self.device.reset(); + } let rdram = self.device.rdram_mut(); @@ -180,10 +242,11 @@ impl Renderer { // Write viewport data (big-endian i16 values → native-endian RDRAM) write_be_bytes_to_rdram(rdram, VIEWPORT_ADDR, &gbi_output.viewport_data); - // Write rspboot to IMEM (big-endian, RSP reads directly) + // Load rspboot into IMEM. It will DMA F3DEX2 text into IMEM and + // data into DMEM, then jump to the microcode entry point. self.device.imem_mut()[..RSPBOOT.len()].copy_from_slice(RSPBOOT); - // Write OSTask to DMEM (big-endian, RSP reads directly with LW) + // Write OSTask to DMEM (big-endian, RSP reads directly with LW). let dmem = self.device.dmem_mut(); let task_base = TASK_DMEM_OFFSET; @@ -221,12 +284,10 @@ impl Renderer { write_be_u32(dmem, task_base + TASK_DATA_SIZE, dl_size); write_be_u32(dmem, task_base + TASK_YIELD_DATA_PTR, 0); - // Decode IMEM and run RSP self.device.decode_imem(); self.device.set_pc(0); self.device.clear_halt(); - - self.device.run().to_vec() + self.warm = true; } } @@ -359,4 +420,60 @@ mod tests { "RSP should produce RDP commands for a single triangle" ); } + + #[test] + fn render_to_streams_to_sink() { + struct CountingSink(usize); + impl rsp::RdpSink for CountingSink { + fn receive_commands(&mut self, commands: &[u32]) { + self.0 += commands.len(); + } + } + + let nodes = vec![NodeData { + triangles: vec![TriangleData { + v0: VertexData { + x: 0, + y: 0, + z: 0, + r: 255, + g: 0, + b: 0, + a: 255, + }, + v1: VertexData { + x: 100, + y: 0, + z: 0, + r: 255, + g: 0, + b: 0, + a: 255, + }, + v2: VertexData { + x: 50, + y: 100, + z: 0, + r: 255, + g: 0, + b: 0, + a: 255, + }, + }], + }]; + + let camera = CameraMatrices { + projection: identity_n64_matrix(), + modelview: identity_n64_matrix(), + }; + + let mut renderer = Renderer::new(); + let mut sink = CountingSink(0); + renderer.render_to(&nodes, &camera, &mut sink); + + assert!( + sink.0 > 0, + "render_to should have streamed RDP commands to the sink" + ); + } } diff --git a/crates/rsp/src/lib.rs b/crates/rsp/src/lib.rs index b80cf20..6c48534 100644 --- a/crates/rsp/src/lib.rs +++ b/crates/rsp/src/lib.rs @@ -24,6 +24,8 @@ pub mod rsp_interface; pub mod su_instructions; pub mod vu_instructions; +pub use rdp::RdpSink; + /// Branch state enum used by the RSP CPU pipeline. #[derive(PartialEq, Copy, Clone)] pub enum BranchStepState { @@ -68,6 +70,11 @@ pub struct Device { pub byte_swap: usize, /// Maximum total cycles before `run()` forcibly halts. pub max_cycles: u64, + /// Active RDP command sink, set only during [`run_with_sink`](Self::run_with_sink). + /// + /// Raw pointer because `run_rdp` is called deep in the RSP execution + /// stack and all intermediate functions take `&mut Device`. + pub(crate) sink: Option<*mut dyn rdp::RdpSink>, } impl Device { @@ -83,6 +90,7 @@ impl Device { mi: Mi { regs: [0; 4] }, byte_swap: 0, max_cycles: DEFAULT_MAX_CYCLES, + sink: None, }; rsp_interface::init(&mut device); rdp::init(&mut device); @@ -99,18 +107,89 @@ impl Device { self.rdp = rdp::Rdp::new(); self.mi = Mi { regs: [0; 4] }; self.byte_swap = 0; + self.sink = None; rsp_interface::init(self); rdp::init(self); } + /// Resets RSP/RDP control state so the device can run again. + /// + /// Unlike [`reset`](Self::reset), this preserves RDRAM, RSP memory + /// (DMEM/IMEM), the decoded instruction cache, lookup tables, and + /// dispatch tables. Only CPU flags, SP registers, and DPC registers + /// are cleared. + pub fn rearm(&mut self) { + // CPU flags + self.rsp.cpu.broken = false; + self.rsp.cpu.halted = false; + self.rsp.cpu.running = false; + self.rsp.cpu.sync_point = false; + self.rsp.cpu.cycle_counter = 0; + self.rsp.cpu.pipeline_full = false; + self.rsp.cpu.branch_state = cpu::BranchState { + state: BranchStepState::Step, + pc: 0, + }; + self.rsp.cpu.last_instruction_type = cpu::InstructionType::Su; + self.rsp.cpu.instruction_type = cpu::InstructionType::Su; + + // SP registers + self.rsp.regs = [0; rsp_interface::SP_REGS_COUNT as usize]; + self.rsp.regs2 = [0; rsp_interface::SP_REGS2_COUNT as usize]; + self.rsp.fifo = [rsp_interface::RspDma { + dir: rsp_interface::DmaDir::None, + length: 0, + memaddr: 0, + dramaddr: 0, + }; 2]; + self.rsp.last_status_value = 0; + self.rsp.run_after_dma = false; + + // DPC/DPS registers + self.rdp.regs_dpc = [0; rdp::DPC_REGS_COUNT as usize]; + self.rdp.regs_dps = [0; rdp::DPS_REGS_COUNT as usize]; + self.rdp.wait_frozen = false; + self.rdp.last_status_value = 0; + self.rdp.collected_commands.clear(); + + self.mi = Mi { regs: [0; 4] }; + self.byte_swap = 0; + self.sink = None; + + // Match the initial state from reset() without regenerating tables + self.rsp.regs[rsp_interface::SP_STATUS_REG as usize] = 1; // HALT + self.rdp.regs_dpc[rdp::DPC_STATUS_REG as usize] |= 1 << 7; // CBUF_READY + } + /// Runs the RSP until it halts or breaks, then returns the collected RDP /// command words. /// - /// The RSP may hit sync points during DMA and DPC operations. This method - /// automatically resumes execution after each sync point, looping until - /// the RSP truly halts or breaks. + /// Commands are buffered in `rdp.collected_commands`. For streaming + /// delivery to a GPU backend, use [`run_with_sink`](Self::run_with_sink). pub fn run(&mut self) -> &[u32] { self.rdp.collected_commands.clear(); + self.run_inner(); + &self.rdp.collected_commands + } + + /// Runs the RSP, streaming RDP commands directly to `sink`. + /// + /// Unlike [`run`](Self::run), this does not buffer commands in + /// `collected_commands`. The RDRAM path is zero-copy — the byte + /// slice is reinterpreted as `&[u32]` and passed straight through. + pub fn run_with_sink(&mut self, sink: &mut dyn rdp::RdpSink) { + // SAFETY: The pointer is cleared before this method returns. The + // transmute erases the borrow lifetime so it can be stored in the + // struct, but run_inner is synchronous and the sink reference is + // valid throughout. + self.sink = unsafe { Some(std::mem::transmute(std::ptr::from_mut(sink))) }; + self.run_inner(); + self.sink = None; + } + + /// RSP execution loop shared by [`run`](Self::run) and + /// [`run_with_sink`](Self::run_with_sink). + fn run_inner(&mut self) { let mut total_cycles: u64 = 0; loop { let batch_cycles = cpu::run(self); @@ -120,7 +199,6 @@ impl Device { break; } } - &self.rdp.collected_commands } /// Mutable access to RDRAM for writing data the RSP will read. diff --git a/crates/rsp/src/rdp.rs b/crates/rsp/src/rdp.rs index 6699fd0..1a581b3 100644 --- a/crates/rsp/src/rdp.rs +++ b/crates/rsp/src/rdp.rs @@ -5,9 +5,20 @@ //! RDP (Reality Display Processor) register handling and command collection. //! -//! Instead of sending commands to a GPU backend (as gopher64 does), this -//! standalone version collects the RDP command words into a `Vec` so -//! the caller can pass them to parallel-rdp or another renderer. +//! By default, RDP command words are collected into a `Vec` (see +//! [`Device::run`](crate::Device::run)). When a [`RdpSink`] is installed via +//! [`Device::run_with_sink`](crate::Device::run_with_sink), commands are +//! streamed directly to the sink — zero-copy for the RDRAM path. + +/// Sink for RDP command words produced during RSP execution. +/// +/// Implementations receive batches of commands as the RSP produces them, +/// rather than waiting for a complete `Vec` at the end. Install a +/// sink via [`Device::run_with_sink`](crate::Device::run_with_sink). +pub trait RdpSink { + /// Receives a batch of RDP command words in native byte order. + fn receive_commands(&mut self, commands: &[u32]); +} pub const DPC_START_REG: u32 = 0; pub const DPC_END_REG: u32 = 1; @@ -122,21 +133,54 @@ pub fn write_regs_dpc(device: &mut crate::Device, address: u64, value: u32, mask } } -/// Collects RDP commands from RDRAM between CURRENT and END registers. +/// Dispatches RDP commands from RDRAM (or DMEM in XBUS mode) to either the +/// installed [`RdpSink`] or the fallback `collected_commands` buffer. +/// +/// The RDRAM path is zero-copy when a sink is present: the byte slice is +/// reinterpreted as `&[u32]` in-place. DPC register masking (`& 0xFFFFF8`) +/// guarantees 8-byte alignment for both CURRENT and END. fn run_rdp(device: &mut crate::Device) { let current = device.rdp.regs_dpc[DPC_CURRENT_REG as usize] as usize; let end = device.rdp.regs_dpc[DPC_END_REG as usize] as usize; - if device.rdp.regs_dpc[DPC_STATUS_REG as usize] & DPC_STATUS_XBUS_DMEM_DMA != 0 { - // XBUS mode: commands come from DMEM/IMEM instead of RDRAM - let mut addr = current & 0xFFF; - while addr < (end & 0xFFF) { - let word = u32::from_be_bytes(device.rsp.mem[addr..addr + 4].try_into().unwrap()); - device.rdp.collected_commands.push(word); - addr += 4; + let is_xbus = device.rdp.regs_dpc[DPC_STATUS_REG as usize] & DPC_STATUS_XBUS_DMEM_DMA != 0; + + if is_xbus { + let start = current & 0xFFF; + let end_addr = end & 0xFFF; + if let Some(sink_ptr) = device.sink { + let mut commands = Vec::new(); + let mut addr = start; + while addr < end_addr { + let word = u32::from_be_bytes(device.rsp.mem[addr..addr + 4].try_into().unwrap()); + commands.push(word); + addr += 4; + } + // SAFETY: sink_ptr is valid for the duration of run_with_sink. + unsafe { (*sink_ptr).receive_commands(&commands) }; + } else { + let mut addr = start; + while addr < end_addr { + let word = u32::from_be_bytes(device.rsp.mem[addr..addr + 4].try_into().unwrap()); + device.rdp.collected_commands.push(word); + addr += 4; + } } + } else if let Some(sink_ptr) = device.sink { + let end_clamped = end.min(device.rdram.mem.len()); + let start_clamped = current.min(end_clamped); + let slice = &device.rdram.mem[start_clamped..end_clamped]; + // SAFETY: DPC registers mask addresses with & 0xFFFFF8 (8-byte aligned). + // Vec allocations are at least pointer-aligned. RDRAM stores u32 + // words in native byte order. + let (prefix, commands, suffix) = unsafe { slice.align_to::() }; + debug_assert!( + prefix.is_empty() && suffix.is_empty(), + "RDRAM slice not u32-aligned: {start_clamped:#x}..{end_clamped:#x}" + ); + // SAFETY: sink_ptr is valid for the duration of run_with_sink. + unsafe { (*sink_ptr).receive_commands(commands) }; } else { - // Normal mode: commands come from RDRAM (stored in native byte order) let mut addr = current; while addr < end { if addr + 4 <= device.rdram.mem.len() { -- 2.51.2