From 897899a5f94ffa61eab6c28bb2377e9578b8dd66 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 13:35:51 +0000 Subject: [PATCH] Sleep instead of spinning while waiting for a staging buffer When the GPU is two frames behind, begin_frame waits for the next staging buffer's MapAsync. It used to spin on ProcessEvents, holding a whole core busy for as long as the GPU took. It now calls ProcessEvents, checks for device loss, then sleeps on a condition variable until the map callback lands or 1 ms passes. The map state also carries a generation, so a callback from a request that shutdown or a rotation already retired can no longer mark the current buffer mapped or unmapped. From upstream patchzyy/Wiicompiled 6f14bde (#244, KartPad batch). Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01Wg7mB8ogCWmp9GH19Uc82B --- aurora-main/lib/gfx/common.cpp | 35 ++++++----- aurora-main/lib/gfx/staging_map.hpp | 60 +++++++++++++++++++ .../tests/renderer_regression_test.cpp | 33 ++++++++++ 3 files changed, 110 insertions(+), 18 deletions(-) create mode 100644 aurora-main/lib/gfx/staging_map.hpp diff --git a/aurora-main/lib/gfx/common.cpp b/aurora-main/lib/gfx/common.cpp index e5e3961..bd897b9 100644 --- a/aurora-main/lib/gfx/common.cpp +++ b/aurora-main/lib/gfx/common.cpp @@ -1,4 +1,5 @@ #include "common.hpp" +#include "staging_map.hpp" #include "../gx/shader_info.hpp" #include "clear.hpp" @@ -139,12 +140,7 @@ wgpu::Buffer g_storageBuffer; constexpr size_t FrameSlotCount = 3; static std::array g_stagingBuffers; static size_t currentStagingBuffer = 0; -enum class BufferMapState { - Unmapped, - Mapping, - Mapped, -}; -static std::atomic s_mappingState{BufferMapState::Unmapped}; +static StagingMapState s_mappingState; static wgpu::Limits g_cachedLimits; // Advanced once per logical frame in the seal prologue, under the renderer GPU mutex and with the // producer blocked, so every later reader sees a value that no longer moves. @@ -1048,7 +1044,7 @@ void initialize() { label.c_str()); } currentStagingBuffer = 0; - s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release); + s_mappingState.reset(); map_staging_buffer(); { @@ -1160,6 +1156,8 @@ void shutdown() { g_uniformBuffer = {}; g_indexBuffer = {}; g_storageBuffer = {}; + // Invalidate outstanding callbacks before releasing their buffers. + s_mappingState.reset(); g_stagingBuffers.fill({}); for (auto& pool : g_resolveSourceSnapshotPools) { pool.entry.reset(); @@ -1178,27 +1176,25 @@ void shutdown() { g_inOffscreen = false; g_frameIndex = UINT32_MAX; currentStagingBuffer = 0; - s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release); } void map_staging_buffer() { - auto expected = BufferMapState::Unmapped; - if (!s_mappingState.compare_exchange_strong(expected, BufferMapState::Mapping, std::memory_order_acq_rel, - std::memory_order_acquire)) { + const auto generation = s_mappingState.request(); + if (generation == 0) { return; } g_stagingBuffers[currentStagingBuffer].MapAsync( wgpu::MapMode::Write, 0, StagingBufferSize, wgpu::CallbackMode::AllowSpontaneous, - [](wgpu::MapAsyncStatus status, wgpu::StringView message) { + [generation](wgpu::MapAsyncStatus status, wgpu::StringView message) { + const auto result = status == wgpu::MapAsyncStatus::Success ? BufferMapState::Mapped : BufferMapState::Unmapped; + if (!s_mappingState.complete(generation, result)) return; if (status == wgpu::MapAsyncStatus::CallbackCancelled || status == wgpu::MapAsyncStatus::Aborted) { Log.warn("Buffer mapping {}: {}", magic_enum::enum_name(status), message); - s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release); return; } ASSERT(status == wgpu::MapAsyncStatus::Success, "Buffer mapping failed: {} {}", magic_enum::enum_name(status), message); - s_mappingState.store(BufferMapState::Mapped, std::memory_order_release); }); } @@ -1208,7 +1204,7 @@ static bool begin_frame_impl(bool clearEfb) { ZoneScopedN("Wait for buffer map"); map_staging_buffer(); while (true) { - const auto mappingState = s_mappingState.load(std::memory_order_acquire); + const auto mappingState = s_mappingState.state(); if (mappingState == BufferMapState::Mapped) { break; } @@ -1224,6 +1220,9 @@ static bool begin_frame_impl(bool clearEfb) { return false; } g_instance.ProcessEvents(); + webgpu::fail_if_device_lost(); + // Sleep until the map callback lands (or 1 ms passes) instead of spinning a core on ProcessEvents. + s_mappingState.wait_for_progress(); } } g_recordingSnapshotSlot = currentStagingBuffer; @@ -1296,12 +1295,12 @@ void abort_frame() noexcept { g_textureUploads.clear(); g_textureUpload.release(); } - if (s_mappingState.load(std::memory_order_acquire) == BufferMapState::Mapped) { + if (s_mappingState.state() == BufferMapState::Mapped) { // Pending interpolation tasks hold raw pointers into the mapped staging // range; they must be dropped before the buffer is unmapped and rotated. gx::drop_pending_frame_interpolation_uniforms(); g_stagingBuffers[currentStagingBuffer].Unmap(); - s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release); + s_mappingState.reset(); currentStagingBuffer = (currentStagingBuffer + 1) % g_stagingBuffers.size(); map_staging_buffer(); } @@ -1741,7 +1740,7 @@ static bool end_batch_impl(const wgpu::CommandEncoder& cmd, bool advanceFrame, g_uniformUploadDestination = nullptr; } g_stagingBuffers[currentStagingBuffer].Unmap(); - s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release); + s_mappingState.reset(); g_stats.drawCallCount = g_drawCallCount; g_stats.mergedDrawCallCount = g_mergedDrawCallCount; g_stats.lastVertSize = writeBuffer(g_verts, g_vertexBuffer, VertexBufferSize, "Vertex"); diff --git a/aurora-main/lib/gfx/staging_map.hpp b/aurora-main/lib/gfx/staging_map.hpp new file mode 100644 index 0000000..59796c3 --- /dev/null +++ b/aurora-main/lib/gfx/staging_map.hpp @@ -0,0 +1,60 @@ +#pragma once + +#include +#include +#include +#include + +namespace aurora::gfx { + +enum class BufferMapState { Unmapped, Mapping, Mapped }; + +// The renderer owns request/reset; Dawn may complete a request on another thread. +// An old callback must never publish readiness for a different staging slot. +class StagingMapState { + mutable std::mutex mutex_; + std::condition_variable changed_; + uint64_t generation_ = 0; + BufferMapState state_ = BufferMapState::Unmapped; + +public: + uint64_t request() { + std::lock_guard lock(mutex_); + if (state_ != BufferMapState::Unmapped) return 0; + state_ = BufferMapState::Mapping; + return ++generation_; + } + + bool complete(uint64_t generation, BufferMapState state) { + { + std::lock_guard lock(mutex_); + if (generation != generation_ || state_ != BufferMapState::Mapping) return false; + state_ = state; + } + changed_.notify_all(); + return true; + } + + void reset() { + { + std::lock_guard lock(mutex_); + ++generation_; + state_ = BufferMapState::Unmapped; + } + changed_.notify_all(); + } + + BufferMapState state() const { + std::lock_guard lock(mutex_); + return state_; + } + + void wait_for_progress() { + std::unique_lock lock(mutex_); + // ProcessEvents is still serviced between waits for implementations that + // need it. A spontaneous completion wakes immediately, without polling. + changed_.wait_for(lock, std::chrono::milliseconds(1), [&] { return state_ != BufferMapState::Mapping; }); + } +}; + +} // namespace aurora::gfx diff --git a/aurora-main/tests/renderer_regression_test.cpp b/aurora-main/tests/renderer_regression_test.cpp index 0a845b5..841a6b5 100644 --- a/aurora-main/tests/renderer_regression_test.cpp +++ b/aurora-main/tests/renderer_regression_test.cpp @@ -1,6 +1,9 @@ #include "gx_test_common.hpp" +#include "gfx/staging_map.hpp" #include "gx/pipeline.hpp" +#include + using aurora::gx::g_gxState; namespace { @@ -84,3 +87,33 @@ TEST_F(GXFifoTest, SingleExpandedPrimitiveCannotMergeWithTriangles) { EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u); EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector{0, 1, 2})); } + +TEST(StagingMapping, RetiredCallbacksCannotPublishAnotherBuffersReadiness) { + using namespace aurora::gfx; + StagingMapState state; + const auto old = state.request(); + EXPECT_EQ(state.request(), 0u); + state.reset(); + const auto current = state.request(); + EXPECT_FALSE(state.complete(old, BufferMapState::Mapped)); + EXPECT_FALSE(state.complete(old, BufferMapState::Unmapped)); + EXPECT_EQ(state.state(), BufferMapState::Mapping); + EXPECT_TRUE(state.complete(current, BufferMapState::Mapped)); + EXPECT_FALSE(state.complete(current, BufferMapState::Unmapped)); + EXPECT_EQ(state.state(), BufferMapState::Mapped); +} + +TEST(StagingMapping, AsyncCompletionWakesWaiters) { + using namespace aurora::gfx; + StagingMapState state; + const auto generation = state.request(); + std::thread callback([&] { + std::this_thread::sleep_for(std::chrono::milliseconds(10)); + state.complete(generation, BufferMapState::Mapped); + }); + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(2); + while (state.state() == BufferMapState::Mapping && std::chrono::steady_clock::now() < deadline) + state.wait_for_progress(); + callback.join(); + EXPECT_EQ(state.state(), BufferMapState::Mapped); +}