From 3e8bdcfd7b44085046e98712bcf820f804cd3253 Mon Sep 17 00:00:00 2001 From: iChris4 Date: Wed, 16 Sep 2026 17:06:03 +0200 Subject: [PATCH] Added Support for Multiplayer Player 1 Immersive View --- Launcher/Directory.Build.props | 2 +- OPENXR.md | 31 ++ aurora-main/include/aurora/aurora.h | 4 + aurora-main/lib/aurora.cpp | 10 +- aurora-main/lib/gfx/common.cpp | 88 +++++- aurora-main/lib/gfx/common.hpp | 2 + aurora-main/lib/gfx/efb_ram_copy.cpp | 43 ++- aurora-main/lib/gfx/efb_ram_copy.hpp | 5 +- aurora-main/lib/gfx/stereo_replay.hpp | 68 +++++ aurora-main/lib/gx/command_processor.cpp | 109 ++++++- aurora-main/lib/gx/pipeline.hpp | 6 + aurora-main/tests/CMakeLists.txt | 8 + aurora-main/tests/efb_ram_lifetime_smoke.cpp | 146 ++++++++++ aurora-main/tests/gx_fifo_test.cpp | 65 +++++ .../tests/stereo_multiplayer_smoke.cpp | 265 ++++++++++++++++++ aurora-main/tests/stereo_replay_test.cpp | 83 ++++++ runtime/CMakeLists.txt | 8 + runtime/src/hle/gx/gx_copy.cpp | 5 +- runtime/src/hle/vi.cpp | 7 +- runtime/src/vr/mkw_vr_instrumentation.cpp | 4 +- runtime/src/vr/mkw_vr_policy.cpp | 13 +- runtime/tests/vr_policy_tests.cpp | 79 ++++++ 22 files changed, 1013 insertions(+), 38 deletions(-) create mode 100644 aurora-main/tests/efb_ram_lifetime_smoke.cpp create mode 100644 aurora-main/tests/stereo_multiplayer_smoke.cpp create mode 100644 runtime/tests/vr_policy_tests.cpp diff --git a/Launcher/Directory.Build.props b/Launcher/Directory.Build.props index d0acf2a..fc69fa1 100644 --- a/Launcher/Directory.Build.props +++ b/Launcher/Directory.Build.props @@ -1,5 +1,5 @@ - 0.2.32 + 0.2.38 diff --git a/OPENXR.md b/OPENXR.md index b5d5f74..323ce5c 100644 --- a/OPENXR.md +++ b/OPENXR.md @@ -66,6 +66,30 @@ image is chosen, so the setting can always be changed back. Eye mirror modes retain the last eye image when a desktop frame has no new XR packet, so they do not alternate with the normal camera. `"none"` also stays black between XR packets. +## Local multiplayer + +During 2-, 3-, and 4-player races, the headset replays Player 1's world in immersive stereo +with head tracking. Keep **F10 > VR > Desktop view** set to **Normal** (`mirror_view = "normal"`) +for the original desktop split-screen layout. No extra multiplayer switch is required. +Menus continue to use the virtual screen. + +Only headset replay filters the other players' viewports and expands Player 1 to each eye. +The desktop split-screen partition (the game's `partition_line` layout, one-pixel textured +picture panes on the split boundaries) and full masks of the other panes are omitted from +the eyes; the desktop image keeps them. +Player-local HUD viewports follow Player 1; shared orthographic overlays keep their full-screen +layout on the virtual screen. Framebuffer effects that sample the desktop split-screen image +are omitted from multiplayer eyes, since those textures contain the other cameras too. +The local-screen count is sealed with each frame, including retained VR interpolation frames, +and a layout change invalidates older XR packets. + +Multiplayer uses Player 1's game camera. The optional first-person relocation and model hiding +remain single-player-only: guest model visibility changes would also affect the desktop players. + +The opt-in `stereo_multiplayer_smoke` D3D12 test reads back both eye images and the desktop EFB +for 1/2/3/4/1-screen transitions with VR interpolation on and off. Actual headset racing still +needs visual validation for course effects, HUD layout, pause/resume, and scene transitions. + **F10 > VR > VR frame interpolation (experimental)** offers **Off, Auto, 72, 90, 120** and applies immediately. `frame_interpolation_fps` stores `0` for Off (the default), `1` for Auto, or the selected rate. The earlier `frame_interpolation = true` checkbox migrates to Auto. @@ -240,6 +264,13 @@ so it can be enabled once Aurora exposes those handles safely. ### Interpolation validation +Probe-sized EFB readbacks retain completed pixels in host memory and publish them only inside +the next compatible `GXCopyTex` call. GPU completion callbacks must not write to guest RAM: +a race restart can reuse a freed probe buffer for `RaceCamera`, and a late 4x4 Z24X8 tile then +turns its rotation fields into NaNs and triggers `triangular.h` / `PPCHalt`. +The optional Windows GPU test `efb_ram_lifetime_smoke` exercises that allocation reuse and +format/size changes. It fails with the former callback write and passes with deferred publication. + The GX tests cover retained transform endpoints with desktop interpolation off and continuous sampling at 72/90/120 Hz. `mkw_frame_interpolation_pacing_tests` covers fixed-rate scheduling, live changes, stalls and configuration migration; `mkw_openxr_replay_tests` exercises swapchain diff --git a/aurora-main/include/aurora/aurora.h b/aurora-main/include/aurora/aurora.h index fb4f823..2fb7468 100644 --- a/aurora-main/include/aurora/aurora.h +++ b/aurora-main/include/aurora/aurora.h @@ -236,6 +236,10 @@ void aurora_end_frame_tagged(uint64_t contentTag); * provider, which cannot know which frame will consume its packet. */ void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]); +// Select Player 1's subview for immersive replay of 2-4 local screens. +// Producer-thread, per-frame metadata, consumed by the next end_frame call. +// One (the default) keeps full-frame replay. Desktop rendering is unaffected. +void aurora_set_stereo_local_player_count(uint32_t count); typedef void (*AuroraFrameWorkerWaitCallback)(); // Called from the producer thread at bounded intervals while Aurora waits for // the asynchronous frame worker. The callback must not enter Aurora. diff --git a/aurora-main/lib/aurora.cpp b/aurora-main/lib/aurora.cpp index ff960cc..9f64af7 100644 --- a/aurora-main/lib/aurora.cpp +++ b/aurora-main/lib/aurora.cpp @@ -100,11 +100,13 @@ struct StereoSceneAnchor { 0.f, 0.f, 1.f, 0.f, }; bool active = false; + uint32_t localPlayerCount = 1; }; // Producer thread only, between aurora_set_stereo_scene_anchor() and the seal // that consumes it. Cleared at every seal so a producer that stops publishing // falls back to the recorded camera instead of freezing on a stale anchor. StereoSceneAnchor g_pendingSceneAnchor; +uint32_t g_pendingStereoLocalPlayerCount = 1; using PresentClock = std::chrono::steady_clock; @@ -1762,6 +1764,7 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u // current_frame() advances inside gfx::end_frame; unsigned wrap maps the // pre-first-frame UINT32_MAX value to logical frame zero. ctx.logicalFrame = gfx::current_frame() + 1; + gfx::set_stereo_local_player_count(sceneAnchor.localPlayerCount); if (const auto stereoInput = request_stereo_frame(ctx.logicalFrame, contentTag)) { ctx.stereoInput = stereoInput; ctx.stereoFrameToken = stereoInput->frameToken; @@ -2246,7 +2249,9 @@ void end_frame(uint64_t contentTag) noexcept { #endif // Claim the anchor published for this frame. Clearing it here is what makes a // producer that stops publishing fall back to the recorded camera. - const StereoSceneAnchor sceneAnchor = g_pendingSceneAnchor; + StereoSceneAnchor sceneAnchor = g_pendingSceneAnchor; + sceneAnchor.localPlayerCount = g_pendingStereoLocalPlayerCount; + g_pendingStereoLocalPlayerCount = 1; g_pendingSceneAnchor = {}; if (!frame_worker_requested()) { end_frame_impl(true, true, contentTag, sceneAnchor); @@ -2378,6 +2383,9 @@ void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]) { aurora::set_stereo_scene_anchor(anchorFromScene); } +void aurora_set_stereo_local_player_count(uint32_t count) { + aurora::g_pendingStereoLocalPlayerCount = count >= 1 && count <= 4 ? count : 1; +} void aurora_set_frame_worker_wait_callback(AuroraFrameWorkerWaitCallback callback) { aurora::g_frameWorkerWaitCallback.store(callback, std::memory_order_release); } diff --git a/aurora-main/lib/gfx/common.cpp b/aurora-main/lib/gfx/common.cpp index b8ffcad..4acabf3 100644 --- a/aurora-main/lib/gfx/common.cpp +++ b/aurora-main/lib/gfx/common.cpp @@ -305,6 +305,7 @@ static void recycle_render_passes(std::vector& passes) noexcept { struct LateStereoUniform { gx::UniformReplayLayout layout; Viewport viewport; + ClipRect displayRegion; Range current; Range previous; std::array eyes; @@ -322,9 +323,16 @@ struct LateStereoData { // Advanced for every upload, including synchronous mid-frame EFB readbacks. static std::atomic_uint64_t g_replayBufferGeneration{0}; static LateStereoData g_pendingLateStereo; +static uint32_t g_stereoLocalPlayerCount = 1; + +void set_stereo_local_player_count(uint32_t count) noexcept { + g_stereoLocalPlayerCount = count >= 1 && count <= 4 ? count : 1; +} + struct SealedFrameData { std::vector passes; LateStereoData stereo; + uint32_t localPlayerCount = 1; }; SealedFrame::SealedFrame() : m_data(std::make_unique()) {} @@ -1396,6 +1404,10 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, LateStereoData* history = nullptr) noexcept { const StereoDisplaySource displaySource = stereo_display_source(g_renderPasses); const ClipRect displayRegion = displaySource.region; + const bool multiplayer = g_stereoLocalPlayerCount > 1; + const auto playerRegion = stereo_replay::player_one_region( + {float(displayRegion.x), float(displayRegion.y), float(displayRegion.width), float(displayRegion.height)}, + g_stereoLocalPlayerCount); // This is the producer-side preparation path; eye replay can query the pure // helper concurrently without touching this diagnostic state. log_stereo_display_source_region(displayRegion, displaySource.foundDisplayCopy); @@ -1403,18 +1415,25 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, // A draw is replayed per eye when it carries the game camera (perspective) or // when it is 2D content the virtual screen is claiming. const auto replayed = [&](const gx::UniformReplayLayout& layout) noexcept { - return layout.perspective || (hudScreen.valid() && !layout.nativeEfbEffect); + return (!multiplayer || !layout.nativeEfbEffect) && + (layout.perspective || (hudScreen.valid() && !layout.nativeEfbEffect)); }; size_t requiredBytes = 0; size_t efbPassCount = 0; size_t perspectiveDrawCount = 0; size_t replayPerspectiveDrawCount = 0; size_t replayHudScreenDrawCount = 0; + stereo_replay::SubviewRect allocationViewport{float(displayRegion.x), float(displayRegion.y), + float(displayRegion.width), float(displayRegion.height)}; for (const auto& pass : g_renderPasses) { if (pass.efbTarget) { ++efbPassCount; } for (const auto& command : pass.commands) { + if (pass.efbTarget && command.type == CommandType::SetViewport) { + const auto& vp = command.data.setViewport; + allocationViewport = {vp.left, vp.top, vp.width, vp.height}; + } if (command.type != CommandType::Draw || command.data.draw.type != ShaderType::GX || !replayed(command.data.draw.gx.uniformReplayLayout)) { continue; @@ -1427,6 +1446,10 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, if (!pass.efbTarget) { continue; } + if (multiplayer && !stereo_replay::replay_player_one_draw(allocationViewport, playerRegion, layout.perspective, + layout.nativeEfbEffect)) { + continue; + } if (layout.perspective) { ++replayPerspectiveDrawCount; } else { @@ -1499,6 +1522,18 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, } auto& draw = command.data.draw.gx; const auto& layout = draw.uniformReplayLayout; + const stereo_replay::SubviewRect viewportRect{drawViewport.left, drawViewport.top, drawViewport.width, + drawViewport.height}; + if (multiplayer && !stereo_replay::replay_player_one_draw(viewportRect, playerRegion, layout.perspective, + layout.nativeEfbEffect)) { + continue; + } + // Pane-local HUD coordinates expand with P1. Shared race/pause overlays + // retain their full-screen layout on the virtual screen. + const bool playerLocal = multiplayer && stereo_replay::subview_contains(playerRegion, viewportRect); + const ClipRect uniformRegion = playerLocal ? ClipRect{int32_t(playerRegion.left), int32_t(playerRegion.top), + int32_t(playerRegion.width), int32_t(playerRegion.height)} + : displayRegion; std::memcpy(sourceUniform.data(), g_uniforms.data() + draw.uniformRange.offset, draw.uniformRange.size); Mat4x4 gameProjection; std::memcpy(&gameProjection, sourceUniform.data() + layout.projectionOffset, sizeof(gameProjection)); @@ -1522,6 +1557,7 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, saved = &history->uniforms.emplace_back(); saved->layout = layout; saved->viewport = drawViewport; + saved->displayRegion = uniformRegion; const auto save = [&](const uint8_t* source, uint32_t size) -> Range { if (size == 0) return {}; @@ -1544,7 +1580,7 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, } const auto& eye = stereoFrame.eyes[eyeIndex]; std::memcpy(eyeUniform.data(), sourceUniform.data(), range.size); - write_stereo_uniform({eyeUniform.data(), range.size}, layout, eye, gameProjection, drawViewport, displayRegion, + write_stereo_uniform({eyeUniform.data(), range.size}, layout, eye, gameProjection, drawViewport, uniformRegion, hudScreen); std::memcpy(uniform.data(), eyeUniform.data(), range.size); } @@ -1681,6 +1717,7 @@ struct RenderInvocation { uint32_t stereoEye = UINT32_MAX; const ReplayTarget* target = nullptr; ClipRect replaySourceRegion{}; + uint32_t localPlayerCount = 1; // Inclusive index of the last pass to replay; -1 replays every pass. int32_t replayLastPass = -1; bool finalize = true; @@ -1857,6 +1894,8 @@ void seal_frame(SealedFrame& out) noexcept { // producer joins the worker's DONE phase before it seals another frame. g_retiredBindGroups.clear(); out.data().stereo = std::move(g_pendingLateStereo); + out.data().localPlayerCount = g_stereoLocalPlayerCount; + g_stereoLocalPlayerCount = 1; auto& passes = out.data().passes; // The previous cycle already recycled these, so this normally just hands the empty vector, its // capacity included, back to the producer. @@ -1928,7 +1967,7 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c } for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) { write_stereo_uniform({bytes + saved.eyes[eye].offset - data.uploadOffset, saved.current.size}, layout, - stereoFrame.eyes[eye], projection, saved.viewport, data.displayRegion, data.hudScreen); + stereoFrame.eyes[eye], projection, saved.viewport, saved.displayRegion, data.hudScreen); } } // Never interpolate in mapped upload memory: write-combined pages make CPU @@ -1961,6 +2000,7 @@ void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, const Ster .stereoEye = eye, .target = &stereoFrame.eyes[eye].target, .replaySourceRegion = displaySource.region, + .localPlayerCount = frame.data().localPlayerCount, .replayLastPass = lastPass, .finalize = finalize, .replayOnlyEfb = true, @@ -2010,6 +2050,12 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec const auto& sourceSize = renderPasses[idx].targetSize; const bool overrideTarget = invocation.target != nullptr && renderPasses[idx].efbTarget; const auto targetSize = overrideTarget ? invocation.target->size : sourceSize; + const bool multiplayer = overrideTarget && invocation.localPlayerCount > 1; + const auto& display = invocation.replaySourceRegion; + const auto playerRegion = stereo_replay::player_one_region( + {float(display.x), float(display.y), float(display.width), float(display.height)}, invocation.localPlayerCount); + stereo_replay::SubviewRect sourceViewport{0.f, 0.f, float(sourceSize.width), float(sourceSize.height)}; + auto sourceScissor = sourceViewport; const int32_t sourceWidth = static_cast(sourceSize.width); const int32_t sourceHeight = static_cast(sourceSize.height); int32_t sourceRegionLeft = 0; @@ -2086,6 +2132,7 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec switch (cmd.type) { case CommandType::SetViewport: { const auto& vp = cmd.data.setViewport; + sourceViewport = {vp.left, vp.top, vp.width, vp.height}; // WebGPU requires 0 <= minDepth <= maxDepth <= 1. vp.znear/vp.zfar are in GX's own distance // terms (0 = near); under UseReversedZ the host depth-buffer storage direction is flipped // (near = 1, far = 0), so this range has to be remapped through 1-x the same way the @@ -2121,6 +2168,7 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec } break; case CommandType::SetScissor: { const auto& sc = cmd.data.setScissor; + sourceScissor = {float(sc.x), float(sc.y), float(sc.width), float(sc.height)}; const auto sourceLeft = std::clamp(sc.x, sourceRegionLeft, sourceRegionRight); const auto sourceTop = std::clamp(sc.y, sourceRegionTop, sourceRegionBottom); const auto sourceRight = std::clamp(sc.x + sc.width, sourceLeft, sourceRegionRight); @@ -2146,6 +2194,18 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec const auto& draw = cmd.data.draw; switch (draw.type) { case ShaderType::GX: { + if (multiplayer && draw.gx.screenRect && + stereo_replay::is_split_screen_furniture( + *draw.gx.screenRect, {float(display.x), float(display.y), float(display.width), float(display.height)}, + invocation.localPlayerCount)) { + break; + } + if (multiplayer && (!stereo_replay::replay_player_one_draw(sourceViewport, playerRegion, + draw.gx.uniformReplayLayout.perspective, + draw.gx.uniformReplayLayout.nativeEfbEffect) || + !stereo_replay::subviews_overlap(sourceScissor, playerRegion))) { + break; + } const gfx::Range* uniformOverride = nullptr; // Only a 2D draw the virtual screen actually claimed carries a stereo // uniform range without being perspective. @@ -2164,21 +2224,31 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec // frame (Mario Kart clips the item roulette that way). Honouring that // rectangle would cut the reprojected element away, so it gets the whole // eye and every other draw gets the game's own rectangle back. - if (!scissorStateKnown || virtualScreenDraw != hudScreenScissor) { - hudScreenScissor = virtualScreenDraw; + const bool fullEyeDraw = virtualScreenDraw || (multiplayer && draw.gx.uniformReplayLayout.perspective); + if (!scissorStateKnown || fullEyeDraw != hudScreenScissor) { + hudScreenScissor = fullEyeDraw; scissorStateKnown = true; - apply_scissor(virtualScreenDraw ? fullTargetScissor : recordedScissor); + apply_scissor(fullEyeDraw ? fullTargetScissor : recordedScissor); } - if (!viewportStateKnown || virtualScreenDraw != hudScreenViewport) { - hudScreenViewport = virtualScreenDraw; + if (!viewportStateKnown || fullEyeDraw != hudScreenViewport) { + hudScreenViewport = fullEyeDraw; viewportStateKnown = true; - apply_viewport(virtualScreenDraw); + apply_viewport(fullEyeDraw); } gx::render(draw.gx, pass, encodeState, renderPasses[idx].requireReadyPipelines, uniformOverride, virtualScreenDraw ? draw.gx.exactScreenDepthPipeline : 0); } break; case ShaderType::Clear: { auto clearDraw = draw.clear; + if (multiplayer) { + const auto& sc = clearDraw.scissor; + if (clearDraw.copyClear || + (clearDraw.useScissor && + !stereo_replay::subviews_overlap({float(sc.x), float(sc.y), float(sc.width), float(sc.height)}, + playerRegion))) { + break; + } + } if (invocation.skipCopyClears && overrideTarget && clearDraw.copyClear && renderPasses[idx].postCopyClear) { // The scissored twin of the attachment-load-op case above: the copy's // EFB reset, rescaled into eye space, covers the whole eye. diff --git a/aurora-main/lib/gfx/common.hpp b/aurora-main/lib/gfx/common.hpp index 83d98b4..493e9a5 100644 --- a/aurora-main/lib/gfx/common.hpp +++ b/aurora-main/lib/gfx/common.hpp @@ -316,6 +316,8 @@ struct StereoReplayFrame { }; void end_frame(const wgpu::CommandEncoder& cmd); +// Set under the renderer mutex immediately before preparing/sealing the frame. +void set_stereo_local_player_count(uint32_t count) noexcept; // Prepares eye-specific uniform copies before unmapping the staging buffer. // Returns false without modifying the mono path when the uniform buffer has // insufficient room for the additional copies. diff --git a/aurora-main/lib/gfx/efb_ram_copy.cpp b/aurora-main/lib/gfx/efb_ram_copy.cpp index 34d507b..c4136f4 100644 --- a/aurora-main/lib/gfx/efb_ram_copy.cpp +++ b/aurora-main/lib/gfx/efb_ram_copy.cpp @@ -60,7 +60,6 @@ struct AsyncSlot { uint64_t bufferSize = 0; uint32_t bytesPerRow = 0; // Latched at encode time and read by the map callback. - void* dest = nullptr; GXTexFmt format = GX_TF_RGBA8; uint32_t width = 0; uint32_t height = 0; @@ -68,6 +67,14 @@ struct AsyncSlot { uint32_t hostHeight = 0; HostPixelOrder order = HostPixelOrder::RGBA; AsyncState state = AsyncState::Idle; + // A guest address is not an allocation lifetime: a scene restart can reuse a + // probe buffer for a camera before this slot's GPU work completes. Callbacks + // therefore retain results here; only a new schedule() may publish to RAM. + std::array completedData{}; + GXTexFmt completedFormat = GX_TF_RGBA8; + uint32_t completedWidth = 0; + uint32_t completedHeight = 0; + size_t completedSize = 0; }; std::vector g_pending; @@ -138,17 +145,18 @@ void complete_async_slot(void* dest, wgpu::MapAsyncStatus status, wgpu::StringVi if (status == wgpu::MapAsyncStatus::Success) { const auto* pixels = static_cast(slot.buffer.GetConstMappedRange(0, slot.bufferSize)); if (pixels != nullptr) { - // Writes guest RAM from the event-queue thread while the guest may be reading it. The only - // consumer min/maxes depth for a fade factor, so a torn tile just mixes two frames' depths. const size_t outputSize = encoded_size(slot.format, slot.width, slot.height); - if (!encode(slot.dest, outputSize, slot.format, slot.width, slot.height, pixels, slot.hostWidth, slot.hostHeight, - slot.bytesPerRow, slot.order)) { + if (!encode(slot.completedData.data(), slot.completedData.size(), slot.format, slot.width, slot.height, pixels, + slot.hostWidth, slot.hostHeight, slot.bytesPerRow, slot.order)) { + slot.completedSize = 0; Log.error("Failed to encode async EFB RAM copy format=0x{:x} size={}x{}", static_cast(slot.format), slot.width, slot.height); + } else { + slot.completedFormat = slot.format; + slot.completedWidth = slot.width; + slot.completedHeight = slot.height; + slot.completedSize = outputSize; } - // Guest RAM written from outside the embedder, so nothing bumps its write generation and a - // texture cached over this range would keep its digest. - notify_guest_write(slot.dest, outputSize); } slot.buffer.Unmap(); } else if (status != wgpu::MapAsyncStatus::CallbackCancelled && status != wgpu::MapAsyncStatus::Aborted) { @@ -201,12 +209,22 @@ void schedule(void* dest, uint32_t width, uint32_t height, GXTexFmt format, Text async = false; } else { g_asyncSlots.try_emplace(dest); - // No readback has landed here yet. 0xff decodes to far Z, which the probe reads as unobstructed - // so a new flare fades in; zero-filled RAM would decode as fully occluded. - std::memset(dest, 0xff, encodedSize); - notify_guest_write(dest, encodedSize); } } + if (async) { + const auto& slot = g_asyncSlots.at(dest); + // schedule runs on the producer inside GXCopyTex, while this destination + // is owned by the copy. Never let a later GPU callback write to it. + if (slot.completedSize == encodedSize && slot.completedFormat == format && slot.completedWidth == width && + slot.completedHeight == height) { + std::memcpy(dest, slot.completedData.data(), encodedSize); + } else { + // No compatible result yet. Far Z makes a new flare fade in instead + // of treating uninitialized RAM as a fully occluded probe. + std::memset(dest, 0xff, encodedSize); + } + notify_guest_write(dest, encodedSize); + } } auto& list = async ? g_asyncPending : g_pending; @@ -395,7 +413,6 @@ void encode_async_downloads(const wgpu::CommandEncoder& encoder) noexcept { }; encoder.CopyTextureToBuffer(&source, &destination, &texture->size); slot.bytesPerRow = bytesPerRow; - slot.dest = pending.dest; slot.format = pending.format; slot.width = pending.width; slot.height = pending.height; diff --git a/aurora-main/lib/gfx/efb_ram_copy.hpp b/aurora-main/lib/gfx/efb_ram_copy.hpp index 6a20279..3f336b9 100644 --- a/aurora-main/lib/gfx/efb_ram_copy.hpp +++ b/aurora-main/lib/gfx/efb_ram_copy.hpp @@ -15,7 +15,10 @@ bool complete_downloads() noexcept; void cancel() noexcept; // Frame-latent readbacks for probe-sized CPU-consumed copies: they ride the frame's own encode and -// land in guest RAM a frame later. Per frame, from the worker: seal, encode, then after_submit. +// retain completed pixels in host memory. The producer publishes a compatible +// result only when schedule() is called again for that destination, while the +// game still owns it as a copy buffer. Per frame, from the worker: seal, encode, +// then after_submit. GPU callbacks must never write to guest RAM. void seal_async_downloads() noexcept; void encode_async_downloads(const wgpu::CommandEncoder& encoder) noexcept; void after_submit() noexcept; diff --git a/aurora-main/lib/gfx/stereo_replay.hpp b/aurora-main/lib/gfx/stereo_replay.hpp index 5565869..24f69e3 100644 --- a/aurora-main/lib/gfx/stereo_replay.hpp +++ b/aurora-main/lib/gfx/stereo_replay.hpp @@ -1,9 +1,77 @@ #pragma once #include +#include namespace aurora::gfx::stereo_replay { +struct SubviewRect { + float left = 0.f; + float top = 0.f; + float width = 0.f; + float height = 0.f; +}; + +// MKW uses a top/bottom split for two screens and quadrants for three/four. +// Coordinates belong to the displayed EFB region, including its crop origin. +inline SubviewRect player_one_region(SubviewRect display, uint32_t players) noexcept { + if (players >= 2 && players <= 4) { + display.height *= 0.5f; + if (players >= 3) { + display.width *= 0.5f; + } + } + return display; +} + +inline bool subviews_overlap(SubviewRect a, SubviewRect b) noexcept { + return a.width > 0.f && a.height > 0.f && b.width > 0.f && b.height > 0.f && a.left < b.left + b.width && + a.left + a.width > b.left && a.top < b.top + b.height && a.top + a.height > b.top; +} + +inline bool subview_contains(SubviewRect outer, SubviewRect inner) noexcept { + // GX viewport jitter and rounding can extend a pane by a fraction of a pixel. + constexpr float tolerance = 1.f; + return inner.width > 0.f && inner.height > 0.f && inner.left >= outer.left - tolerance && + inner.top >= outer.top - tolerance && inner.left + inner.width <= outer.left + outer.width + tolerance && + inner.top + inner.height <= outer.top + outer.height + tolerance; +} + +// Shared orthographic overlays may span the display. World geometry must belong +// wholly to P1; otherwise another camera could be expanded into the same eye. +// EFB effects sample the desktop's multi-camera image, so they cannot be reused. +inline bool replay_player_one_draw(SubviewRect viewport, SubviewRect playerRegion, bool perspective, + bool nativeEfbEffect) noexcept { + return !nativeEfbEffect && + (perspective ? subview_contains(playerRegion, viewport) : subviews_overlap(playerRegion, viewport)); +} + +// Split-screen furniture is often geometry in a full-display orthographic +// viewport, not a separate viewport. MKW draws its partition as the +// partition_line layout: yoko_line (800x1) and tate_line (1x800) picture panes +// centred on the display and sampling a pattern texture. Recognize only +// rectangles/lines at the split boundaries and complete masks of other +// players' panes, whether textured or not. Full-frame fades and small HUD +// backgrounds must remain visible. +inline bool is_split_screen_furniture(SubviewRect bounds, SubviewRect display, uint32_t players) noexcept { + if (players < 2 || players > 4 || display.width <= 0.f || display.height <= 0.f) + return false; + const float x = (bounds.left - display.left) / display.width; + const float y = (bounds.top - display.top) / display.height; + const float w = bounds.width / display.width; + const float h = bounds.height / display.height; + constexpr float tolerance = 0.008f; // Up to a few native EFB pixels of inset/jitter. + const auto near = [](float a, float b) { return std::abs(a - b) <= tolerance; }; + if (h <= tolerance && w >= 0.45f && near(y + h * 0.5f, 0.5f)) + return true; + if (players >= 3 && w <= tolerance && h >= 0.45f && near(x + w * 0.5f, 0.5f)) + return true; + if (players == 2) + return near(x, 0.f) && near(y, 0.5f) && near(w, 1.f) && near(h, 0.5f); + return near(w, 0.5f) && near(h, 0.5f) && + ((near(x, 0.5f) && (near(y, 0.f) || near(y, 0.5f))) || (near(x, 0.f) && near(y, 0.5f))); +} + // An OpenXR eye supplies the shape of its asymmetric frustum, but the sealed // GX draw already contains the depth mapping adjusted for that draw's GX // viewport and Aurora's reversed-Z convention. Replacing the complete matrix diff --git a/aurora-main/lib/gx/command_processor.cpp b/aurora-main/lib/gx/command_processor.cpp index 61b04da..7a28be0 100644 --- a/aurora-main/lib/gx/command_processor.cpp +++ b/aurora-main/lib/gx/command_processor.cpp @@ -1861,7 +1861,7 @@ static u32 calculate_last_vtx_size(GXVtxFmt fmt) { static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange, uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature, - bool interpolationIdentityActive); + bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride); // The per-draw geometry signature, matrix-usage mask and draw-identity hashes exist purely to feed frame interpolation // (build_uniform consumes them only after its `frame_interpolation_fps() == 0` early-out). @@ -1891,6 +1891,104 @@ static uint32_t matrix_index_prefix_size(GXVtxFmt fmt) noexcept { return size; } +// Screen-space bounds of a simple orthographic rectangle or line, textured or +// not: MKW's split-screen partition is a layout picture pane (a one-pixel quad +// sampling a pattern texture), so texture use cannot disqualify a candidate. +// The geometry rules in is_split_screen_furniture keep HUD art visible. +static std::optional screen_rect(GXPrimitive prim, GXVtxFmt fmt, + const uint8_t* vertices, uint16_t count, + uint32_t stride) noexcept { + if (!aurora::stereo_frame_provider_active() || g_gxState.projType != GX_ORTHOGRAPHIC || + !((count == 4 && (prim == GX_QUADS || prim == GX_TRIANGLESTRIP || prim == GX_TRIANGLEFAN)) || + (count == 2 && prim == GX_LINES))) + return {}; + const auto& projection = g_gxState.proj; + if (!gfx::stereo_replay::is_orthographic_projection(projection)) + return {}; + const auto& attr = g_gxState.vtxFmts[fmt].attrs[GX_VA_POS]; + const uint32_t components = attr.cnt == GX_POS_XY ? 2 : 3; + const uint32_t componentBytes = comp_type_size(GX_VA_POS, attr.type); + if (componentBytes == 0) + return {}; + const uint32_t offset = matrix_index_prefix_size(fmt); + std::array, 4> points{}; + float left = INFINITY, top = INFINITY, right = -INFINITY, bottom = -INFINITY; + for (uint32_t i = 0; i < count; ++i) { + const auto* vertex = vertices + i * stride; + const auto* position = vertex + offset; + bool bigEndian = true; + const auto type = g_gxState.vtxDesc[GX_VA_POS]; + if (type == GX_INDEX8 || type == GX_INDEX16) { + const uint32_t index = type == GX_INDEX8 ? *position : read_u16(position, true); + const auto& array = g_gxState.arrays[GX_VA_POS]; + const size_t start = size_t(index) * array.stride; + if (array.data == nullptr || start + components * componentBytes > array.size) + return {}; + position = static_cast(array.data) + start; + bigEndian = !array.le; + } else if (type != GX_DIRECT || offset + components * componentBytes > stride) { + return {}; + } + std::array point{}; + for (uint32_t c = 0; c < components; ++c) { + const auto* value = position + c * componentBytes; + switch (attr.type) { + case GX_U8: + point[c] = *value; + break; + case GX_S8: + point[c] = static_cast(*value); + break; + case GX_U16: + point[c] = read_u16(value, bigEndian); + break; + case GX_S16: + point[c] = static_cast(read_u16(value, bigEndian)); + break; + case GX_F32: + point[c] = read_f32(value, bigEndian); + break; + default: + return {}; + } + if (attr.type != GX_F32) + point[c] = std::ldexp(point[c], -int(attr.frac)); + } + const uint32_t matrixIndex = + g_gxState.vtxDesc[GX_VA_PNMTXIDX] == GX_DIRECT ? vertex[0] / 3u : g_gxState.currentPnMtx; + if (matrixIndex >= g_gxState.pnMtx.size()) + return {}; + const auto& matrix = g_gxState.pnMtx[matrixIndex].pos; + const auto transform = [](const Vec4& row, const std::array& p) { + return row[0] * p[0] + row[1] * p[1] + row[2] * p[2] + row[3]; + }; + const std::array view{transform(matrix.m0, point), transform(matrix.m1, point), + transform(matrix.m2, point)}; + const auto& vp = g_gxState.renderViewport; + const float x = vp.left + (transform(projection.m0, view) + 1.f) * vp.width * 0.5f; + const float y = vp.top + (1.f - transform(projection.m1, view)) * vp.height * 0.5f; + if (!std::isfinite(x) || !std::isfinite(y)) + return {}; + points[i] = {x, y}; + left = std::min(left, x); + right = std::max(right, x); + top = std::min(top, y); + bottom = std::max(bottom, y); + } + // A diagonal/rotated HUD polygon's bounding box is not a screen mask. + uint32_t corners = 0; + for (uint32_t i = 0; i < count; ++i) { + const auto& p = points[i]; + if ((std::abs(p[0] - left) > 0.01f && std::abs(p[0] - right) > 0.01f) || + (std::abs(p[1] - top) > 0.01f && std::abs(p[1] - bottom) > 0.01f)) + return {}; + corners |= 1u << ((std::abs(p[0] - right) < 0.01f ? 1 : 0) + (std::abs(p[1] - bottom) < 0.01f ? 2 : 0)); + } + if ((count == 4 && corners != 15) || (count == 2 && right - left > 0.01f && bottom - top > 0.01f)) + return {}; + return gfx::stereo_replay::SubviewRect{left, top, right - left, bottom - top}; +} + static HashType draw_geometry_signature(GXVtxFmt fmt, const uint8_t* vertices, uint16_t vtxCount, uint32_t vtxStride) noexcept { Hasher hasher; @@ -2155,7 +2253,7 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{}; handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature, interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0, - interpolationIdentityActive); + interpolationIdentityActive, vertices, vtxSize); return true; } @@ -2189,7 +2287,7 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi pos += totalVtxBytes; // Try to merge with previous draw call - if (!g_gxState.stateDirty) + if (!g_gxState.stateDirty && !(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC)) LIKELY { auto* lastDraw = gfx::get_last_draw_command(); // Only if the previous draw call was a single instance draw (no lines/points handling) @@ -2223,13 +2321,13 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{}; handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature, interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0, - interpolationIdentityActive); + interpolationIdentityActive, vertices, vtxSize); return true; } static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange, uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature, - bool interpolationIdentityActive) { + bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride) { ZoneScoped; // GX_CULL_ALL rasterizes nothing on hardware - no color, no depth. if (g_gxState.cullMode == GX_CULL_ALL && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS) @@ -2320,6 +2418,7 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g .instanceCount = instanceCount, .bindGroups = bindGroups, .dstAlpha = pipelineState.dstAlpha, + .screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride), }); g_gxState.stateDirty = false; } diff --git a/aurora-main/lib/gx/pipeline.hpp b/aurora-main/lib/gx/pipeline.hpp index 3353ff5..e5567a0 100644 --- a/aurora-main/lib/gx/pipeline.hpp +++ b/aurora-main/lib/gx/pipeline.hpp @@ -2,6 +2,8 @@ #include "../gfx/common.hpp" #include "shader_info.hpp" +#include "../gfx/stereo_replay.hpp" +#include namespace aurora::gx { struct DrawData { @@ -21,6 +23,10 @@ struct DrawData { uint32_t instanceCount; GXBindGroups bindGroups; uint32_t dstAlpha; + // Valid only for simple orthographic rectangles/lines (textured or not). + // Recorded before merging so VR can omit desktop split masks without + // modifying GX. + std::optional screenRect; }; constexpr uint32_t GXPipelineConfigVersion = 20; diff --git a/aurora-main/tests/CMakeLists.txt b/aurora-main/tests/CMakeLists.txt index 0ff3763..0917cd6 100644 --- a/aurora-main/tests/CMakeLists.txt +++ b/aurora-main/tests/CMakeLists.txt @@ -7,6 +7,14 @@ if (AURORA_GPU_SMOKE_TESTS AND AURORA_ENABLE_GX AND WIN32) target_include_directories(stereo_frame_worker_smoke PRIVATE ../lib) target_link_libraries(stereo_frame_worker_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi dawn::dawncpp_headers) + add_executable(stereo_multiplayer_smoke stereo_multiplayer_smoke.cpp) + target_include_directories(stereo_multiplayer_smoke PRIVATE ../lib) + target_link_libraries(stereo_multiplayer_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi + dawn::dawncpp_headers) + add_executable(efb_ram_lifetime_smoke efb_ram_lifetime_smoke.cpp) + target_include_directories(efb_ram_lifetime_smoke PRIVATE ../lib) + target_link_libraries(efb_ram_lifetime_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi + dawn::dawncpp_headers) endif () if (NOT TARGET gtest) diff --git a/aurora-main/tests/efb_ram_lifetime_smoke.cpp b/aurora-main/tests/efb_ram_lifetime_smoke.cpp new file mode 100644 index 0000000..281cd1a --- /dev/null +++ b/aurora-main/tests/efb_ram_lifetime_smoke.cpp @@ -0,0 +1,146 @@ +// Opt-in GPU regression for a depth-probe allocation reused during a scene +// restart. A late map callback must not overwrite the new camera allocation. +#include +#include +#include "gfx/efb_ram_copy.hpp" +#include "gfx/efb_ram_encoder.hpp" +#include "webgpu/gpu.hpp" + +#include +#include +#include +#include +#include +#include + +using namespace aurora; +namespace { +constexpr uint8_t kCameraByte = 0x35; +std::array guest; + +gfx::TextureHandle MakeTexture(uint32_t width, uint32_t height, std::array pixel) { + auto texture = gfx::new_render_texture(width, height, GX_TF_RGBA8, "Probe lifetime test"); + if (texture->format == wgpu::TextureFormat::BGRA8Unorm) + std::swap(pixel[0], pixel[2]); + std::array pixels; + for (size_t i = 0; i < width * height; ++i) + std::copy(pixel.begin(), pixel.end(), pixels.begin() + i * 4); + const wgpu::TexelCopyTextureInfo destination{.texture = texture->texture}; + const wgpu::TexelCopyBufferLayout layout{.bytesPerRow = width * 4, .rowsPerImage = height}; + webgpu::g_queue.WriteTexture(&destination, pixels.data(), width * height * 4, &layout, &texture->size); + return texture; +} + +void SubmitProbe(const gfx::TextureHandle& texture, GXTexFmt format, uint32_t width, uint32_t height) { + gfx::efb_ram::schedule(guest.data(), width, height, format, texture); + gfx::efb_ram::seal_async_downloads(); + auto encoder = webgpu::g_device.CreateCommandEncoder(); + gfx::efb_ram::encode_async_downloads(encoder); + auto commands = encoder.Finish(); + webgpu::g_queue.Submit(1, &commands); + // The guest frees the probe and constructs a camera at the same address, + // before the readback callback is registered. + guest.fill(kCameraByte); + gfx::efb_ram::after_submit(); +} + +bool CameraIntact() { + return std::all_of(guest.begin(), guest.end(), [](uint8_t byte) { return byte == kCameraByte; }); +} + +bool AwaitProbe(const gfx::TextureHandle& texture, GXTexFmt format, uint32_t width, uint32_t height, + const std::array& pixel) { + const size_t size = gfx::efb_ram::encoded_size(format, width, height); + std::array pixels{}, expected{}; + for (size_t i = 0; i < width * height; ++i) + std::copy(pixel.begin(), pixel.end(), pixels.begin() + i * 4); + if (!gfx::efb_ram::encode(expected.data(), size, format, width, height, pixels.data(), width, height, width * 4, + gfx::efb_ram::HostPixelOrder::RGBA)) + return false; + + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(5); + do { + // Give the real GPU completion callback a chance to run while the old + // destination belongs to the camera, then verify it remains untouched. + webgpu::g_instance.ProcessEvents(); + std::this_thread::sleep_for(std::chrono::milliseconds(10)); + if (!CameraIntact()) { + std::fprintf(stderr, "Late GPU readback overwrote the reused camera allocation\n"); + return false; + } + // This explicit copy owns the destination again and may receive the result. + gfx::efb_ram::schedule(guest.data(), width, height, format, texture); + const bool arrived = std::equal(expected.begin(), expected.begin() + size, guest.begin()); + const bool canary = + std::all_of(guest.begin() + size, guest.end(), [](uint8_t byte) { return byte == kCameraByte; }); + gfx::efb_ram::cancel(); + guest.fill(kCameraByte); + if (!canary) + return false; + if (arrived) + return true; + } while (std::chrono::steady_clock::now() < deadline); + std::fprintf(stderr, "Completed probe was not delivered by the next compatible copy\n"); + return false; +} + +bool Run() { + const std::array pixel{0xff, 0xfa, 0xb6, 0xff}; + auto large = MakeTexture(8, 8, pixel); + auto small = MakeTexture(4, 4, pixel); + SubmitProbe(large, GX_TF_Z24X8, 8, 8); + if (!AwaitProbe(large, GX_TF_Z24X8, 8, 8, pixel)) + return false; + std::puts("Reused camera allocation survives late 256-byte probe: PASS"); + + // Same address, smaller allocation and different encoding: the retained + // 256-byte result must not escape the new 32-byte destination. + gfx::efb_ram::schedule(guest.data(), 4, 4, GX_TF_Z16, small); + const bool fallback = std::all_of(guest.begin(), guest.begin() + 32, [](uint8_t b) { return b == 0xff; }); + const bool canary = std::all_of(guest.begin() + 32, guest.end(), [](uint8_t b) { return b == kCameraByte; }); + gfx::efb_ram::cancel(); + if (!fallback || !canary) + return false; + std::puts("Reused address rejects incompatible retained format and size: PASS"); + + SubmitProbe(small, GX_TF_Z24X8, 4, 4); + if (!AwaitProbe(small, GX_TF_Z24X8, 4, 4, pixel)) + return false; + std::puts("Crash-dump 64-byte depth pattern delivered only during a new copy: PASS"); + SubmitProbe(small, GX_TF_Z16, 4, 4); + if (!AwaitProbe(small, GX_TF_Z16, 4, 4, pixel)) + return false; + std::puts("Changed probe format delivers compatible pixels without touching canaries: PASS"); + return true; +} +} // namespace + +int main(int argc, char** argv) { + std::setvbuf(stdout, nullptr, _IONBF, 0); + std::filesystem::create_directories("efb-lifetime-cache"); + AuroraConfig config{}; + config.appName = "Aurora EFB lifetime validation"; + config.userPath = "."; + config.cachePath = "efb-lifetime-cache"; + config.desiredBackend = BACKEND_D3D12; + config.windowWidth = 160; + config.windowHeight = 120; + config.hasWindowPosition = true; + config.windowPosX = config.windowPosY = -30000; + config.xrInterop = true; + config.logLevel = LOG_WARNING; + config.logCallback = [](AuroraLogLevel, const char* module, const char* message, unsigned length) { + std::fprintf(stderr, "%s: %.*s\n", module, int(length), message); + }; + aurora_initialize(argc, argv, &config); + std::puts("GPU initialized"); + aurora_begin_frame(); + GXInit(nullptr, 0); + std::puts("Starting readback lifetime checks"); + const bool passed = Run(); + std::puts("Readback lifetime checks finished"); + aurora_end_frame(); + aurora_quiesce_frame_worker(); + aurora_shutdown(); + return passed ? 0 : 1; +} diff --git a/aurora-main/tests/gx_fifo_test.cpp b/aurora-main/tests/gx_fifo_test.cpp index f6ff4fd..b0c5f45 100644 --- a/aurora-main/tests/gx_fifo_test.cpp +++ b/aurora-main/tests/gx_fifo_test.cpp @@ -2269,6 +2269,71 @@ TEST_F(GXFifoTest, MergedDrawOffsetsCachedTopologyWithoutJoiningPrimitives) { EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector{3, 4, 5})); } +TEST_F(GXFifoTest, OrthographicQuadRecordsScreenRectForVrFurniture) { + // MKW draws its split-screen partition with the partition_line layout: a + // one-pixel picture pane sampling a pattern texture in a full-display + // orthographic viewport. VR replay drops it by its recorded screen + // rectangle. This covers the FIFO decode of that rectangle. The harness + // stubs populate_pipeline_config, so whether texture use disqualifies a + // rectangle is exercised by stereo_multiplayer_smoke instead. + aurora::gfx::testing::use_real_vertex_format_helpers(true); + aurora::gfx::testing::use_draw_command_tracking(true); + + aurora::Mat4x4 proj{}; + proj.m0[0] = 2.0f / 640.0f; + proj.m0[3] = -1.0f; + proj.m1[1] = 2.0f / 480.0f; + proj.m1[3] = -1.0f; + proj.m2[2] = -1.0f; + proj.m3[3] = 1.0f; + GXSetProjection(&proj, GX_ORTHOGRAPHIC); + GXSetViewport(0.0f, 0.0f, 640.0f, 480.0f, 0.0f, 1.0f); + GXSetScissor(0, 0, 640, 480); + aurora::Mat3x4 identity{}; + identity.m0[0] = identity.m1[1] = identity.m2[2] = 1.0f; + GXLoadPosMtxImm(&identity, GX_PNMTX0); + GXSetCurrentMtx(GX_PNMTX0); + + GXClearVtxDesc(); + GXSetVtxDesc(GX_VA_POS, GX_DIRECT); + GXSetVtxDesc(GX_VA_CLR0, GX_DIRECT); + GXSetVtxDesc(GX_VA_TEX0, GX_DIRECT); + GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_POS, GX_POS_XYZ, GX_F32, 0); + GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_CLR0, GX_CLR_RGBA, GX_RGBA8, 0); + GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_TEX0, GX_TEX_ST, GX_F32, 0); + GXSetNumChans(1); + GXSetNumTexGens(1); + GXSetTexCoordGen(GX_TEXCOORD0, GX_TG_MTX2x4, GX_TG_TEX0, GX_IDENTITY); + GXSetNumTevStages(1); + GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD0, GX_TEXMAP0, GX_COLOR0A0); + GXSetTevOp(GX_TEVSTAGE0, GX_MODULATE); + alignas(32) static const u8 pattern[8 * 8 * 2]{}; + GXTexObj obj{}; + GXInitTexObj(&obj, pattern, 8, 8, GX_TF_RGB565, GX_CLAMP, GX_CLAMP, GX_FALSE); + GXLoadTexObj(&obj, GX_TEXMAP0); + + // yoko_line: full width, one pixel tall, centred vertically. + const float corners[4][2]{{0.0f, 239.5f}, {640.0f, 239.5f}, {640.0f, 240.5f}, {0.0f, 240.5f}}; + GXBegin(GX_QUADS, GX_VTXFMT0, 4); + for (const auto& corner : corners) { + GXPosition3f32(corner[0], corner[1], 0.0f); + GXColor4u8(0, 0, 0, 255); + GXTexCoord2f32(corner[0] / 640.0f, corner[1] > 240.0f ? 1.0f : 0.0f); + } + GXEnd(); + decode_fifo(flush_and_capture()); + + const auto* draw = aurora::gfx::get_last_draw_command(); + ASSERT_NE(draw, nullptr); + ASSERT_TRUE(draw->screenRect.has_value()); + EXPECT_NEAR(draw->screenRect->left, 0.0f, 0.01f); + EXPECT_NEAR(draw->screenRect->top, 239.5f, 0.01f); + EXPECT_NEAR(draw->screenRect->width, 640.0f, 0.01f); + EXPECT_NEAR(draw->screenRect->height, 1.0f, 0.01f); + EXPECT_TRUE( + aurora::gfx::stereo_replay::is_split_screen_furniture(*draw->screenRect, {0.0f, 0.0f, 640.0f, 480.0f}, 2)); +} + TEST_F(GXFifoTest, TexBufferSize_UsesExactLinearPcFormatSizes) { EXPECT_EQ(GXGetTexBufferSize(8, 4, GX_TF_R8_PC, GX_FALSE, 0), 32u); EXPECT_EQ(GXGetTexBufferSize(8, 4, GX_TF_RGBA8_PC, GX_FALSE, 0), 128u); diff --git a/aurora-main/tests/stereo_multiplayer_smoke.cpp b/aurora-main/tests/stereo_multiplayer_smoke.cpp new file mode 100644 index 0000000..14d2374 --- /dev/null +++ b/aurora-main/tests/stereo_multiplayer_smoke.cpp @@ -0,0 +1,265 @@ +// Opt-in D3D12 readback test: P1 is red, P2 green, P3 blue, P4 yellow. +// Both eyes must be entirely P1 while the original EFB keeps every pane. +#include +#include +#include +#include +#include "stereo.hpp" +#include "gfx/common.hpp" + +#include +#include +#include + +using namespace aurora::webgpu; +namespace { +// Eyes match the EFB width so a one-pixel-class divider on the virtual screen +// still covers eye pixels; a 160x120 eye never rasterized it. +constexpr uint32_t kWidth = 640, kHeight = 480, kEfbWidth = 640, kEfbHeight = 528, kPitch = 2560; +constexpr uint64_t kBytes = kPitch * kEfbHeight; +std::array readbacks; +std::array formats; +uint64_t token = 0; +uint32_t copies = 0; +bool Provide(uint32_t, AuroraStereoFrame* frame, void*) { + *frame = {}; + frame->frameToken = ++token; + frame->contentTag = 42; + for (auto& eye : frame->eyes) { + eye.width = kWidth; + eye.height = kHeight; + eye.projection[0] = eye.projection[5] = 1; + eye.projection[10] = eye.projection[11] = eye.projection[14] = -1; + eye.viewFromCenter[0] = eye.viewFromCenter[5] = eye.viewFromCenter[10] = 1; + } + return true; +} +bool Encode(wgpu::CommandEncoder& encoder, const aurora::stereo::SinkFrame& frame, void*) noexcept { + for (uint32_t i = 0; i < 3; ++i) { + const auto texture = i < 2 ? *frame.eyes[i].texture : present_source().texture; + formats[i] = texture.GetFormat(); + const wgpu::TexelCopyTextureInfo source{.texture = texture}; + const wgpu::TexelCopyBufferInfo destination{.layout = {.bytesPerRow = kPitch, .rowsPerImage = kEfbHeight}, + .buffer = readbacks[i]}; + const wgpu::Extent3D size{i < 2 ? kWidth : kEfbWidth, i < 2 ? kHeight : kEfbHeight, 1}; + encoder.CopyTextureToBuffer(&source, &destination, &size); + } + ++copies; + return true; +} +void Draw(uint32_t players) { + // GX depth occupies [-w, 0], unlike OpenGL's [-w, +w]. + Mtx44 projection{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 0, -1}, {0, 0, -1, 0}}; + Mtx transform{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, -2}}; + GXSetProjection(projection, GX_PERSPECTIVE); + GXSetCurrentMtx(GX_PNMTX0); + GXLoadPosMtxImm(transform, GX_PNMTX0); + GXClearVtxDesc(); + GXSetVtxDesc(GX_VA_POS, GX_DIRECT); + GXSetVtxDesc(GX_VA_CLR0, GX_DIRECT); + GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_POS, GX_POS_XYZ, GX_F32, 0); + GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_CLR0, GX_CLR_RGBA, GX_RGBA8, 0); + GXSetNumTexGens(0); + GXSetNumChans(1); + GXSetChanCtrl(GX_COLOR0A0, GX_FALSE, GX_SRC_REG, GX_SRC_VTX, GX_LIGHT_NULL, GX_DF_NONE, GX_AF_NONE); + GXSetNumTevStages(1); + GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR0A0); + GXSetTevColorIn(GX_TEVSTAGE0, GX_CC_ZERO, GX_CC_ZERO, GX_CC_ZERO, GX_CC_RASC); + GXSetTevAlphaIn(GX_TEVSTAGE0, GX_CA_ZERO, GX_CA_ZERO, GX_CA_ZERO, GX_CA_RASA); + GXSetTevColorOp(GX_TEVSTAGE0, GX_TEV_ADD, GX_TB_ZERO, GX_CS_SCALE_1, GX_TRUE, GX_TEVPREV); + GXSetTevAlphaOp(GX_TEVSTAGE0, GX_TEV_ADD, GX_TB_ZERO, GX_CS_SCALE_1, GX_TRUE, GX_TEVPREV); + GXSetZMode(GX_FALSE, GX_ALWAYS, GX_FALSE); + GXSetAlphaCompare(GX_ALWAYS, 0, GX_AOP_AND, GX_ALWAYS, 0); + GXSetCullMode(GX_CULL_NONE); + GXSetBlendMode(GX_BM_NONE, GX_BL_ONE, GX_BL_ZERO, GX_LO_COPY); + GXSetColorUpdate(GX_TRUE); + GXSetAlphaUpdate(GX_TRUE); + const GXColor colors[]{{255, 0, 0, 255}, {0, 255, 0, 255}, {0, 0, 255, 255}, {255, 255, 0, 255}}; + for (uint32_t p = 0; p < players; ++p) { + const uint32_t width = players >= 3 ? kEfbWidth / 2 : kEfbWidth; + const uint32_t height = players >= 2 ? kEfbHeight / 2 : kEfbHeight; + const uint32_t x = players >= 3 ? (p % 2) * width : 0; + const uint32_t y = players >= 3 ? (p / 2) * height : p * height; + GXSetViewport(float(x), float(y), float(width), float(height), 0, 1); + GXSetScissor(x, y, width, height); + GXBegin(GX_QUADS, GX_VTXFMT0, 4); + GXPosition3f32(-2, -2, 0); + GXColor4u8(colors[p].r, colors[p].g, colors[p].b, 255); + GXPosition3f32(2, -2, 0); + GXColor4u8(colors[p].r, colors[p].g, colors[p].b, 255); + GXPosition3f32(2, 2, 0); + GXColor4u8(colors[p].r, colors[p].g, colors[p].b, 255); + GXPosition3f32(-2, 2, 0); + GXColor4u8(colors[p].r, colors[p].g, colors[p].b, 255); + GXEnd(); + } + if (players == 1) + return; + // MKW's separators and per-pane backing quads can share one full-screen + // orthographic viewport. Viewport filtering alone must not admit them in VR. + Mtx44 ortho{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 0, -0.5f}, {0, 0, 0, 1}}; + Mtx identity{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, 1, 0}}; + GXSetProjection(ortho, GX_ORTHOGRAPHIC); + GXLoadPosMtxImm(identity, GX_PNMTX0); + GXSetViewport(0, 0, kEfbWidth, kEfbHeight, 0, 1); + GXSetScissor(0, 0, kEfbWidth, kEfbHeight); + const auto rect = [](float left, float bottom, float right, float top) { + GXBegin(GX_QUADS, GX_VTXFMT0, 4); + for (const auto& p : + std::array, 4>{{{left, bottom}, {right, bottom}, {right, top}, {left, top}}}) { + GXPosition3f32(p[0], p[1], 0); + GXColor4u8(0, 0, 0, 255); + } + GXEnd(); + }; + if (players >= 3) + rect(0, 0, 1, 1); // The reported black quadrant. + else + rect(-1, -1, 1, 0); + if (players >= 3) { + GXSetLineWidth(6, GX_TO_ZERO); + GXBegin(GX_LINES, GX_VTXFMT0, 2); + GXPosition3f32(0, -1, 0); + GXColor4u8(0, 0, 0, 255); + GXPosition3f32(0, 1, 0); + GXColor4u8(0, 0, 0, 255); + GXEnd(); + } + // MKW's partition_line layout draws the divider as one-pixel picture panes + // that sample a pattern texture, so the divider here is a textured quad too: + // texture use alone must not keep it out of the furniture filter. + alignas(32) static const uint8_t blackTexels[8 * 8 * 2]{}; // RGB565 zero: opaque black. + GXTexObj divider{}; + GXInitTexObj(÷r, blackTexels, 8, 8, GX_TF_RGB565, GX_CLAMP, GX_CLAMP, GX_FALSE); + GXLoadTexObj(÷r, GX_TEXMAP0); + GXSetVtxDesc(GX_VA_TEX0, GX_DIRECT); + GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_TEX0, GX_TEX_ST, GX_F32, 0); + GXSetNumTexGens(1); + GXSetTexCoordGen(GX_TEXCOORD0, GX_TG_MTX2x4, GX_TG_TEX0, GX_IDENTITY); + GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD0, GX_TEXMAP0, GX_COLOR0A0); + GXSetTevColorIn(GX_TEVSTAGE0, GX_CC_ZERO, GX_CC_ZERO, GX_CC_ZERO, GX_CC_TEXC); + GXSetTevAlphaIn(GX_TEVSTAGE0, GX_CA_ZERO, GX_CA_ZERO, GX_CA_ZERO, GX_CA_TEXA); + GXBegin(GX_QUADS, GX_VTXFMT0, 4); + for (const auto& p : + std::array, 4>{{{-1, -0.006f}, {1, -0.006f}, {1, 0.006f}, {-1, 0.006f}}}) { + GXPosition3f32(p[0], p[1], 0); + GXColor4u8(255, 255, 255, 255); + GXTexCoord2f32(p[0] * 0.5f + 0.5f, p[1] > 0 ? 1.f : 0.f); + } + GXEnd(); +} +bool Check(uint32_t players) { + bool okay = copies != 0; + for (uint32_t i = 0; i < 3; ++i) { + wgpu::MapAsyncStatus status{}; + const auto future = readbacks[i].MapAsync(wgpu::MapMode::Read, 0, kBytes, wgpu::CallbackMode::WaitAnyOnly, + [&](wgpu::MapAsyncStatus result, wgpu::StringView) { status = result; }); + if (g_instance.WaitAny(future, 5'000'000'000) != wgpu::WaitStatus::Success || + status != wgpu::MapAsyncStatus::Success) { + return false; + } + const auto* bytes = static_cast(readbacks[i].GetConstMappedRange()); + for (uint32_t quadrant = 0; quadrant < 4; ++quadrant) { + const uint32_t width = i < 2 ? kWidth : kEfbWidth; + const uint32_t height = i < 2 ? kHeight : kEfbHeight; + const uint32_t x = (quadrant % 2) * width / 2 + width / 4; + const uint32_t y = (quadrant / 2) * height / 2 + height / 4; + uint32_t player = i < 2 || players == 1 ? 0 : players == 2 ? quadrant / 2 : quadrant; + if (player >= players) + continue; + const auto* pixel = bytes + y * kPitch + x * 4; + const bool bgra = formats[i] == wgpu::TextureFormat::BGRA8Unorm; + const uint8_t r = pixel[bgra ? 2 : 0], g = pixel[1], b = pixel[bgra ? 0 : 2]; + const bool mask = i == 2 && players > 1 && (players == 2 ? quadrant >= 2 : quadrant == 1); + const bool match = r == (!mask && (player == 0 || player == 3) ? 255 : 0) && + g == (!mask && (player == 1 || player == 3) ? 255 : 0) && + b == (!mask && player == 2 ? 255 : 0); + if (!match) + std::fprintf(stderr, "%uP target %u quadrant %u expected player %u, got %u,%u,%u\n", players, i, quadrant, + player + 1, r, g, b); + okay &= match; + } + if (i < 2) { + // Include the divider locations; quadrant-center samples alone miss them. + for (uint32_t y = 4; y < kHeight - 4; ++y) { + for (uint32_t x = 4; x < kWidth - 4; ++x) { + const auto* pixel = bytes + y * kPitch + x * 4; + const bool bgra = formats[i] == wgpu::TextureFormat::BGRA8Unorm; + if (pixel[bgra ? 2 : 0] != 255 || pixel[1] != 0 || pixel[bgra ? 0 : 2] != 0) { + std::fprintf(stderr, "%uP eye %u has split-screen overlay at %u,%u\n", players, i, x, y); + okay = false; + y = kHeight; + break; + } + } + } + } else if (players > 1) { + const auto* divider = bytes + (kEfbHeight / 2) * kPitch + (kEfbWidth / 4) * 4; + if (divider[0] != 0 || divider[1] != 0 || divider[2] != 0) { + std::fprintf(stderr, "Desktop divider was removed\n"); + okay = false; + } + } + readbacks[i].Unmap(); + } + return okay; +} +} // namespace + +int main(int argc, char** argv) { + std::filesystem::create_directories("stereo-multiplayer-cache"); + AuroraConfig config{}; + config.appName = "Aurora multiplayer VR validation"; + config.userPath = "."; + config.cachePath = "stereo-multiplayer-cache"; + config.desiredBackend = BACKEND_D3D12; + config.windowWidth = kWidth; + config.windowHeight = kHeight; + config.hasWindowPosition = true; + config.windowPosX = config.windowPosY = -30000; + config.xrInterop = true; + config.logLevel = LOG_WARNING; + config.logCallback = [](AuroraLogLevel, const char* module, const char* message, unsigned length) { + std::fprintf(stderr, "%s: %.*s\n", module, int(length), message); + }; + aurora_initialize(argc, argv, &config); + aurora_set_skip_unready_pipelines(false); + aurora_begin_frame(); + GXInit(nullptr, 0); + const wgpu::BufferDescriptor descriptor{.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst, + .size = kBytes}; + for (auto& buffer : readbacks) + buffer = g_device.CreateBuffer(&descriptor); + aurora_set_stereo_frame_provider(Provide, nullptr); + aurora::gfx::set_stereo_hud_screen(true, 2.f, 1.f); + aurora::stereo::set_sink(Encode, nullptr); + bool okay = true; + for (bool interpolate : {false, true}) { + aurora_set_stereo_frame_interpolation(interpolate); + for (uint32_t players : {1u, 2u, 3u, 4u, 1u}) { + for (uint32_t frame = 0; frame < 3; ++frame) { + aurora_update(); + if (!aurora_begin_frame()) + return 2; + Draw(players); + // Deliberately omit the setter for 1P to exercise per-frame reset. + if (players > 1) + aurora_set_stereo_local_player_count(players); + aurora_end_frame_tagged(42); + aurora_begin_frame(); + aurora_wait_for_frame_worker(); + } + const bool passed = Check(players); + okay &= passed; + std::printf("%u players, interpolation %s: %s\n", players, interpolate ? "on" : "off", passed ? "PASS" : "FAIL"); + } + } + aurora_set_stereo_frame_interpolation(false); + aurora_quiesce_frame_worker(); + aurora_set_stereo_frame_provider(nullptr, nullptr); + aurora::stereo::set_sink(nullptr, nullptr); + for (auto& buffer : readbacks) + buffer = nullptr; + aurora_shutdown(); + return okay ? 0 : 1; +} diff --git a/aurora-main/tests/stereo_replay_test.cpp b/aurora-main/tests/stereo_replay_test.cpp index 8e929a0..4ed7a2b 100644 --- a/aurora-main/tests/stereo_replay_test.cpp +++ b/aurora-main/tests/stereo_replay_test.cpp @@ -8,6 +8,89 @@ namespace aurora::gfx::stereo_replay { namespace { +TEST(StereoReplayTest, SplitFurnitureIsRecognizedInFullDisplayCoordinates) { + const SubviewRect display{16.f, 8.f, 1280.f, 912.f}; + for (uint32_t players : {2u, 3u, 4u}) { + EXPECT_TRUE(is_split_screen_furniture({16.f, 462.f, 1280.f, 4.f}, display, players)); + EXPECT_FALSE(is_split_screen_furniture(display, display, players)); // Race fade. + EXPECT_FALSE(is_split_screen_furniture({80.f, 60.f, 100.f, 40.f}, display, players)); // HUD backing. + EXPECT_FALSE(is_split_screen_furniture(player_one_region(display, players), display, players)); + } + EXPECT_TRUE(is_split_screen_furniture({16.f, 464.f, 1280.f, 456.f}, display, 2)); + for (uint32_t players : {3u, 4u}) { + EXPECT_TRUE(is_split_screen_furniture({656.f, 8.f, 0.f, 912.f}, display, players)); + EXPECT_TRUE(is_split_screen_furniture({658.f, 10.f, 636.f, 452.f}, display, players)); + EXPECT_TRUE(is_split_screen_furniture({16.f, 464.f, 640.f, 456.f}, display, players)); + EXPECT_TRUE(is_split_screen_furniture({656.f, 464.f, 640.f, 456.f}, display, players)); + } + EXPECT_FALSE(is_split_screen_furniture({656.f, 8.f, 640.f, 456.f}, display, 1)); + EXPECT_FALSE(is_split_screen_furniture({656.f, 8.f, 0.f, 912.f}, display, 2)); +} + +TEST(StereoReplayTest, PartitionLineLayoutPanesAreFurniture) { + // MKW's partition_line.brlyt draws yoko_line (800x1) and tate_line (1x800) + // picture panes centred on the display; both extend past a 4:3 root and are + // clipped by the display copy, so only the centre line and thickness matter. + const SubviewRect display{0.f, 0.f, 893.f, 456.f}; + const SubviewRect yoko{46.5f, 227.5f, 800.f, 1.f}; + const SubviewRect tate{446.f, -172.f, 1.f, 800.f}; + EXPECT_TRUE(is_split_screen_furniture(yoko, display, 2)); + EXPECT_FALSE(is_split_screen_furniture(tate, display, 2)); + for (uint32_t players : {3u, 4u}) { + EXPECT_TRUE(is_split_screen_furniture(yoko, display, players)); + EXPECT_TRUE(is_split_screen_furniture(tate, display, players)); + } + // A textured pane of the same shape elsewhere is HUD art, not furniture. + EXPECT_FALSE(is_split_screen_furniture({46.5f, 100.f, 800.f, 1.f}, display, 2)); + EXPECT_FALSE(is_split_screen_furniture({46.5f, 227.5f, 300.f, 1.f}, display, 2)); +} + +TEST(StereoReplayTest, MultiplayerSelectsOnlyPlayerOneWorld) { + const SubviewRect display{12.f, 8.f, 640.f, 456.f}; + for (uint32_t count : {2u, 3u, 4u}) { + const auto player = player_one_region(display, count); + EXPECT_FLOAT_EQ(player.left, display.left); + EXPECT_FLOAT_EQ(player.top, display.top); + EXPECT_FLOAT_EQ(player.width, count == 2 ? 640.f : 320.f); + EXPECT_FLOAT_EQ(player.height, 228.f); + EXPECT_TRUE(replay_player_one_draw(player, player, true, false)); + auto opponent = player; + opponent.top += player.height; + EXPECT_FALSE(replay_player_one_draw(opponent, player, true, false)); + EXPECT_FALSE(replay_player_one_draw(opponent, player, false, false)); + if (count >= 3) { + opponent = player; + opponent.left += player.width; + EXPECT_FALSE(replay_player_one_draw(opponent, player, true, false)); + EXPECT_FALSE(replay_player_one_draw(opponent, player, false, false)); + opponent.top += player.height; // P4 / unused fourth quadrant in 3P. + EXPECT_FALSE(replay_player_one_draw(opponent, player, true, false)); + } + EXPECT_FALSE(replay_player_one_draw(display, player, true, false)); + EXPECT_TRUE(replay_player_one_draw(display, player, false, false)); + EXPECT_FALSE(replay_player_one_draw(player, player, false, true)); + EXPECT_FALSE(replay_player_one_draw(display, player, false, true)); + const auto remap = make_hud_ndc_remap(player.left, player.top, player.width, player.height, player.left, player.top, + player.width, player.height); + EXPECT_FLOAT_EQ(remap.scaleX, 1.f); + EXPECT_FLOAT_EQ(remap.scaleY, 1.f); + EXPECT_FLOAT_EQ(remap.offsetX, 0.f); + EXPECT_FLOAT_EQ(remap.offsetY, 0.f); + } + const auto single = player_one_region(display, 1); + EXPECT_FLOAT_EQ(single.width, display.width); + EXPECT_FLOAT_EQ(single.height, display.height); +} + +TEST(StereoReplayTest, MultiplayerRejectsEmptyAndNonOverlappingScissors) { + const SubviewRect player{0.f, 0.f, 320.f, 228.f}; + EXPECT_FALSE(subviews_overlap(player, {320.f, 0.f, 320.f, 228.f})); + EXPECT_FALSE(subviews_overlap(player, {0.f, 228.f, 640.f, 228.f})); + EXPECT_FALSE(subviews_overlap(player, {10.f, 10.f, 0.f, 10.f})); + EXPECT_TRUE(subviews_overlap(player, {10.f, 10.f, 20.f, 20.f})); + EXPECT_TRUE(subview_contains(player, {0.f, -0.5f, 320.f, 228.f})); +} + TEST(StereoReplayTest, EyeFrustumPreservesGameDepthMapping) { const Mat4x4 game{ {10.0f, 11.0f, 12.0f, 13.0f}, diff --git a/runtime/CMakeLists.txt b/runtime/CMakeLists.txt index 608efea..8735e13 100644 --- a/runtime/CMakeLists.txt +++ b/runtime/CMakeLists.txt @@ -356,6 +356,14 @@ target_include_directories(mkw_vr_player_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR target_compile_features(mkw_vr_player_tests PRIVATE cxx_std_17) add_test(NAME mkw_vr_player_tests COMMAND mkw_vr_player_tests) +add_executable(mkw_vr_policy_tests tests/vr_policy_tests.cpp src/vr/mkw_vr_policy.cpp) +target_include_directories(mkw_vr_policy_tests PRIVATE "${CMAKE_CURRENT_LIST_DIR}/include") +target_compile_features(mkw_vr_policy_tests PRIVATE cxx_std_17) +if(CMAKE_CXX_COMPILER_ID MATCHES "Clang|GNU") + target_compile_options(mkw_vr_policy_tests PRIVATE -ffast-math) +endif() +add_test(NAME mkw_vr_policy_tests COMMAND mkw_vr_policy_tests) + if(MKW_ENABLE_OPENXR AND MKW_PLATFORM_WINDOWS) add_executable(mkw_openxr_replay_tests tests/openxr_d3d12_replay_tests.cpp src/vr/openxr_d3d12.cpp) diff --git a/runtime/src/hle/gx/gx_copy.cpp b/runtime/src/hle/gx/gx_copy.cpp index 5631a01..734125e 100644 --- a/runtime/src/hle/gx/gx_copy.cpp +++ b/runtime/src/hle/gx/gx_copy.cpp @@ -147,8 +147,9 @@ extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) { // EFB exactly once, matching Dolphin's ConvertEFBRectangle path. GXSetTexCopySrc(rawSrcLeft, rawSrcTop, rawSrcWidth, rawSrcHeight); // EFB copies stay GPU-only except probe-sized ones (e.g. the 4x4 lens-flare depth probe), - // which Aurora reads back asynchronously via efb_ram::schedule and land in guest RAM a frame - // later. RISK: copies above the probe threshold, or on the offscreen list, are not + // which Aurora reads back asynchronously and publishes during the next copy to that buffer. + // GPU callbacks retain pixels in host memory so a scene restart cannot receive a late write + // into a freed/reused allocation. RISK: copies above the probe threshold, or on the offscreen list, are not // auto-downloaded, so guest reads see stale RAM; call aurora_flush_efb_copies_to_ram if a // copy needs reading back. GXCopyTex(GuestToHostPtr(da), (GXBool)c); diff --git a/runtime/src/hle/vi.cpp b/runtime/src/hle/vi.cpp index f4ce750..2443bd5 100644 --- a/runtime/src/hle/vi.cpp +++ b/runtime/src/hle/vi.cpp @@ -615,8 +615,11 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) { // asynchronous worker may ask for an XR packet after the guest has already // begun the next frame, so immersive replay is accepted only when both // tags match. - const uint64_t vrContentTag = mkw::vr::MkwVRPolicyGetSnapshot().content_tag; - aurora_end_frame_tagged(vrContentTag); + const auto vrPolicy = mkw::vr::MkwVRPolicyGetSnapshot(); + aurora_set_stereo_local_player_count( + vrPolicy.presentation == mkw::vr::VRPresentationMode::ImmersiveRace + ? vrPolicy.scene.local_player_count : 1); + aurora_end_frame_tagged(vrPolicy.content_tag); if (paceThisFrame) { PaceToRetraceBoundary(paceDeadline); std::lock_guard lock(g_viMutex); diff --git a/runtime/src/vr/mkw_vr_instrumentation.cpp b/runtime/src/vr/mkw_vr_instrumentation.cpp index b80c4eb..3c0f4f3 100644 --- a/runtime/src/vr/mkw_vr_instrumentation.cpp +++ b/runtime/src/vr/mkw_vr_instrumentation.cpp @@ -168,7 +168,9 @@ extern "C" void MkwVRObserveTranslatedFunctionEntry(uint32_t address, case kRaceCameraUpdate: { const uint32_t camera_address = context != nullptr ? context->gpr[3] : 0; ObserveCamera(frame, camera_address); - PublishObservedCamera(frame, camera_address); + if (camera_address == FirstCamera(frame)) { + PublishObservedCamera(frame, camera_address); + } break; } case kScnMgrRaceDraw: { diff --git a/runtime/src/vr/mkw_vr_policy.cpp b/runtime/src/vr/mkw_vr_policy.cpp index 8f3a17f..84c0c36 100644 --- a/runtime/src/vr/mkw_vr_policy.cpp +++ b/runtime/src/vr/mkw_vr_policy.cpp @@ -86,12 +86,13 @@ VRPresentationMode SelectPresentation(const PolicyState& state) noexcept { return VRPresentationMode::Desktop; } - // A virtual screen is the fail-safe for menus, split-screen, and any + // A virtual screen is the fail-safe for menus and any // incomplete instrumentation. It preserves the unmodified render path. if ((state.available_bindings & kMkwVRRequiredImmersiveBindings) != kMkwVRRequiredImmersiveBindings || !state.config.immersive_races || state.scene.mode != VRSceneMode::Race || - state.scene.local_player_count != 1 || !IsFiniteCamera(state.camera) || + (state.scene.local_player_count < 1 || state.scene.local_player_count > 4) || + !IsFiniteCamera(state.camera) || !ObservationsAreCoherent(state.scene, state.camera)) { return VRPresentationMode::VirtualScreen; } @@ -111,7 +112,8 @@ VRPresentationMode SelectStablePresentation(const PolicyState& state) noexcept { if ((state.available_bindings & kMkwVRRequiredImmersiveBindings) != kMkwVRRequiredImmersiveBindings || !state.config.immersive_races || state.scene.mode != VRSceneMode::Race || - state.scene.local_player_count != 1 || !IsFiniteCamera(state.camera)) { + (state.scene.local_player_count < 1 || state.scene.local_player_count > 4) || + !IsFiniteCamera(state.camera)) { return VRPresentationMode::VirtualScreen; } @@ -211,6 +213,11 @@ void MkwVRPolicySetAvailableBindings(uint32_t bindings) noexcept { void MkwVRPolicyPublishScene(const MkwVRSceneObservation& scene) noexcept { std::lock_guard lock(g_policy_mutex); + // A packet for a different split layout must not consume retained content, + // even when both layouts use immersive presentation. + if (scene.local_player_count != g_policy.scene.local_player_count) { + AdvanceSafetyGeneration(g_policy); + } ApplyPolicyMutation([&] { if (scene.mode != VRSceneMode::Race || g_policy.scene.mode != VRSceneMode::Race) { // Never carry a camera sample across a menu/replay-to-race transition. diff --git a/runtime/tests/vr_policy_tests.cpp b/runtime/tests/vr_policy_tests.cpp new file mode 100644 index 0000000..aac7bbe --- /dev/null +++ b/runtime/tests/vr_policy_tests.cpp @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: GPL-3.0-or-later +#include "vr/mkw_vr_policy.h" + +#include +#include + +using namespace mkw::vr; + +int main() { + int failures = 0; + const auto check = [&](bool condition, const char* message) { + if (!condition) { + std::cerr << message << '\n'; + ++failures; + } + }; + MkwVRPolicyReset(); + MkwVRPolicyConfig config{}; + config.enabled = true; + MkwVRPolicyConfigure(config); + MkwVRPolicySetSessionActive(true); + MkwVRPolicySetAvailableBindings(kMkwVRRequiredImmersiveBindings); + MkwVRSceneObservation scene{VRSceneMode::Race, 1, 10}; + MkwVRCameraObservation camera{}; + camera.view_from_world = {1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0}; + camera.guest_camera_address = 0x81000000; + camera.guest_frame_index = 10; + camera.valid = true; + MkwVRPolicyPublishScene(scene); + MkwVRPolicyPublishRaceCamera(camera); + uint64_t previous_tag = 0; + for (uint32_t players : {1u, 2u, 3u, 4u, 2u, 1u}) { + scene.local_player_count = players; + MkwVRPolicyPublishScene(scene); + const auto snapshot = MkwVRPolicyGetSnapshot(); + check(snapshot.presentation == VRPresentationMode::ImmersiveRace, "1-4 players should be immersive"); + check(snapshot.content_tag != previous_tag, "layout changes must invalidate retained packets"); + previous_tag = snapshot.content_tag; + MkwVRPolicyPublishScene(scene); + MkwVRPolicyPublishRaceCamera(camera); + check(MkwVRPolicyGetSnapshot().content_tag == previous_tag, "steady frames must keep the same tag"); + } + camera.guest_frame_index = 11; + MkwVRPolicyPublishRaceCamera(camera); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::ImmersiveRace, + "adjacent camera publication is coherent"); + camera.guest_frame_index = 12; + MkwVRPolicyPublishRaceCamera(camera); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::VirtualScreen, "stale camera fails safe"); + camera.guest_frame_index = 10; + MkwVRPolicyPublishRaceCamera(camera); + for (uint32_t players : {0u, 5u, UINT32_MAX}) { + scene.local_player_count = players; + MkwVRPolicyPublishScene(scene); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::VirtualScreen, "invalid counts fail safe"); + } + scene.local_player_count = 4; + MkwVRPolicyPublishScene(scene); + const uint32_t nan_bits = 0x7fc00000; + std::memcpy(&camera.view_from_world[0], &nan_bits, sizeof(nan_bits)); + MkwVRPolicyPublishRaceCamera(camera); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::VirtualScreen, "NaN camera fails under fast math"); + camera.view_from_world[0] = 1; + MkwVRPolicyPublishRaceCamera(camera); + MkwVRPolicySetAvailableBindings(MkwVRBindingRaceCamera); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::VirtualScreen, "partial hooks fail safe"); + MkwVRPolicySetAvailableBindings(kMkwVRRequiredImmersiveBindings); + scene.mode = VRSceneMode::FrontEnd; + MkwVRPolicyPublishScene(scene); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::VirtualScreen, "menus use virtual screen"); + scene.mode = VRSceneMode::Race; + MkwVRPolicyPublishScene(scene); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::VirtualScreen, "race entry needs fresh camera"); + MkwVRPolicyPublishRaceCamera(camera); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::ImmersiveRace, "race resumes with fresh camera"); + MkwVRPolicySetSessionActive(false); + check(MkwVRPolicyGetSnapshot().presentation == VRPresentationMode::Desktop, "inactive session uses desktop"); + return failures == 0 ? 0 : 1; +}