Sleep instead of spinning while waiting for a staging buffer

When the GPU is two frames behind, begin_frame waits for the next
staging buffer's MapAsync. It used to spin on ProcessEvents, holding a
whole core busy for as long as the GPU took. It now calls ProcessEvents,
checks for device loss, then sleeps on a condition variable until the
map callback lands or 1 ms passes.

The map state also carries a generation, so a callback from a request
that shutdown or a rotation already retired can no longer mark the
current buffer mapped or unmapped.

From upstream patchzyy/Wiicompiled 6f14bde (#244, KartPad batch).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Wg7mB8ogCWmp9GH19Uc82B
This commit is contained in:
Claude committed 2026-10-05 13:35:51 +00:00
1 parent c8220c0c11
commit 897899a5f9
3 files changed
+110 -18

No files matched your search

+17 -18
View File
@@ -1,4 +1,5 @@
#include "common.hpp"
#include "staging_map.hpp"
#include "../gx/shader_info.hpp"
#include "clear.hpp"
@@ -139,12 +140,7 @@ wgpu::Buffer g_storageBuffer;
constexpr size_t FrameSlotCount = 3;
static std::array<wgpu::Buffer, FrameSlotCount> g_stagingBuffers;
static size_t currentStagingBuffer = 0;
enum class BufferMapState {
Unmapped,
Mapping,
Mapped,
};
static std::atomic s_mappingState{BufferMapState::Unmapped};
static StagingMapState s_mappingState;
static wgpu::Limits g_cachedLimits;
// Advanced once per logical frame in the seal prologue, under the renderer GPU mutex and with the
// producer blocked, so every later reader sees a value that no longer moves.
@@ -1048,7 +1044,7 @@ void initialize() {
label.c_str());
}
currentStagingBuffer = 0;
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
s_mappingState.reset();
map_staging_buffer();
{
@@ -1160,6 +1156,8 @@ void shutdown() {
g_uniformBuffer = {};
g_indexBuffer = {};
g_storageBuffer = {};
// Invalidate outstanding callbacks before releasing their buffers.
s_mappingState.reset();
g_stagingBuffers.fill({});
for (auto& pool : g_resolveSourceSnapshotPools) {
pool.entry.reset();
@@ -1178,27 +1176,25 @@ void shutdown() {
g_inOffscreen = false;
g_frameIndex = UINT32_MAX;
currentStagingBuffer = 0;
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
}
void map_staging_buffer() {
auto expected = BufferMapState::Unmapped;
if (!s_mappingState.compare_exchange_strong(expected, BufferMapState::Mapping, std::memory_order_acq_rel,
std::memory_order_acquire)) {
const auto generation = s_mappingState.request();
if (generation == 0) {
return;
}
g_stagingBuffers[currentStagingBuffer].MapAsync(
wgpu::MapMode::Write, 0, StagingBufferSize, wgpu::CallbackMode::AllowSpontaneous,
[](wgpu::MapAsyncStatus status, wgpu::StringView message) {
[generation](wgpu::MapAsyncStatus status, wgpu::StringView message) {
const auto result = status == wgpu::MapAsyncStatus::Success ? BufferMapState::Mapped : BufferMapState::Unmapped;
if (!s_mappingState.complete(generation, result)) return;
if (status == wgpu::MapAsyncStatus::CallbackCancelled || status == wgpu::MapAsyncStatus::Aborted) {
Log.warn("Buffer mapping {}: {}", magic_enum::enum_name(status), message);
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
return;
}
ASSERT(status == wgpu::MapAsyncStatus::Success, "Buffer mapping failed: {} {}", magic_enum::enum_name(status),
message);
s_mappingState.store(BufferMapState::Mapped, std::memory_order_release);
});
}
@@ -1208,7 +1204,7 @@ static bool begin_frame_impl(bool clearEfb) {
ZoneScopedN("Wait for buffer map");
map_staging_buffer();
while (true) {
const auto mappingState = s_mappingState.load(std::memory_order_acquire);
const auto mappingState = s_mappingState.state();
if (mappingState == BufferMapState::Mapped) {
break;
}
@@ -1224,6 +1220,9 @@ static bool begin_frame_impl(bool clearEfb) {
return false;
}
g_instance.ProcessEvents();
webgpu::fail_if_device_lost();
// Sleep until the map callback lands (or 1 ms passes) instead of spinning a core on ProcessEvents.
s_mappingState.wait_for_progress();
}
}
g_recordingSnapshotSlot = currentStagingBuffer;
@@ -1296,12 +1295,12 @@ void abort_frame() noexcept {
g_textureUploads.clear();
g_textureUpload.release();
}
if (s_mappingState.load(std::memory_order_acquire) == BufferMapState::Mapped) {
if (s_mappingState.state() == BufferMapState::Mapped) {
// Pending interpolation tasks hold raw pointers into the mapped staging
// range; they must be dropped before the buffer is unmapped and rotated.
gx::drop_pending_frame_interpolation_uniforms();
g_stagingBuffers[currentStagingBuffer].Unmap();
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
s_mappingState.reset();
currentStagingBuffer = (currentStagingBuffer + 1) % g_stagingBuffers.size();
map_staging_buffer();
}
@@ -1741,7 +1740,7 @@ static bool end_batch_impl(const wgpu::CommandEncoder& cmd, bool advanceFrame,
g_uniformUploadDestination = nullptr;
}
g_stagingBuffers[currentStagingBuffer].Unmap();
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
s_mappingState.reset();
g_stats.drawCallCount = g_drawCallCount;
g_stats.mergedDrawCallCount = g_mergedDrawCallCount;
g_stats.lastVertSize = writeBuffer(g_verts, g_vertexBuffer, VertexBufferSize, "Vertex");
+60
View File
@@ -0,0 +1,60 @@
#pragma once
#include <chrono>
#include <condition_variable>
#include <cstdint>
#include <mutex>
namespace aurora::gfx {
enum class BufferMapState { Unmapped, Mapping, Mapped };
// The renderer owns request/reset; Dawn may complete a request on another thread.
// An old callback must never publish readiness for a different staging slot.
class StagingMapState {
mutable std::mutex mutex_;
std::condition_variable changed_;
uint64_t generation_ = 0;
BufferMapState state_ = BufferMapState::Unmapped;
public:
uint64_t request() {
std::lock_guard lock(mutex_);
if (state_ != BufferMapState::Unmapped) return 0;
state_ = BufferMapState::Mapping;
return ++generation_;
}
bool complete(uint64_t generation, BufferMapState state) {
{
std::lock_guard lock(mutex_);
if (generation != generation_ || state_ != BufferMapState::Mapping) return false;
state_ = state;
}
changed_.notify_all();
return true;
}
void reset() {
{
std::lock_guard lock(mutex_);
++generation_;
state_ = BufferMapState::Unmapped;
}
changed_.notify_all();
}
BufferMapState state() const {
std::lock_guard lock(mutex_);
return state_;
}
void wait_for_progress() {
std::unique_lock lock(mutex_);
// ProcessEvents is still serviced between waits for implementations that
// need it. A spontaneous completion wakes immediately, without polling.
changed_.wait_for(lock, std::chrono::milliseconds(1), [&] { return state_ != BufferMapState::Mapping; });
}
};
} // namespace aurora::gfx
@@ -1,6 +1,9 @@
#include "gx_test_common.hpp"
#include "gfx/staging_map.hpp"
#include "gx/pipeline.hpp"
#include <thread>
using aurora::gx::g_gxState;
namespace {
@@ -84,3 +87,33 @@ TEST_F(GXFifoTest, SingleExpandedPrimitiveCannotMergeWithTriangles) {
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
}
TEST(StagingMapping, RetiredCallbacksCannotPublishAnotherBuffersReadiness) {
using namespace aurora::gfx;
StagingMapState state;
const auto old = state.request();
EXPECT_EQ(state.request(), 0u);
state.reset();
const auto current = state.request();
EXPECT_FALSE(state.complete(old, BufferMapState::Mapped));
EXPECT_FALSE(state.complete(old, BufferMapState::Unmapped));
EXPECT_EQ(state.state(), BufferMapState::Mapping);
EXPECT_TRUE(state.complete(current, BufferMapState::Mapped));
EXPECT_FALSE(state.complete(current, BufferMapState::Unmapped));
EXPECT_EQ(state.state(), BufferMapState::Mapped);
}
TEST(StagingMapping, AsyncCompletionWakesWaiters) {
using namespace aurora::gfx;
StagingMapState state;
const auto generation = state.request();
std::thread callback([&] {
std::this_thread::sleep_for(std::chrono::milliseconds(10));
state.complete(generation, BufferMapState::Mapped);
});
const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(2);
while (state.state() == BufferMapState::Mapping && std::chrono::steady_clock::now() < deadline)
state.wait_for_progress();
callback.join();
EXPECT_EQ(state.state(), BufferMapState::Mapped);
}