mirror of
https://github.com/mitch030504/Wiicompiled_VR_Frame.git
synced 2026-10-06 07:00:30 +02:00
Sleep instead of spinning while waiting for a staging buffer
When the GPU is two frames behind, begin_frame waits for the next staging buffer's MapAsync. It used to spin on ProcessEvents, holding a whole core busy for as long as the GPU took. It now calls ProcessEvents, checks for device loss, then sleeps on a condition variable until the map callback lands or 1 ms passes. The map state also carries a generation, so a callback from a request that shutdown or a rotation already retired can no longer mark the current buffer mapped or unmapped. From upstream patchzyy/Wiicompiled 6f14bde (#244, KartPad batch). Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Wg7mB8ogCWmp9GH19Uc82B
This commit is contained in:
3 files changed
+110
-18
No files matched your search
@@ -1,4 +1,5 @@
|
||||
#include "common.hpp"
|
||||
#include "staging_map.hpp"
|
||||
#include "../gx/shader_info.hpp"
|
||||
|
||||
#include "clear.hpp"
|
||||
@@ -139,12 +140,7 @@ wgpu::Buffer g_storageBuffer;
|
||||
constexpr size_t FrameSlotCount = 3;
|
||||
static std::array<wgpu::Buffer, FrameSlotCount> g_stagingBuffers;
|
||||
static size_t currentStagingBuffer = 0;
|
||||
enum class BufferMapState {
|
||||
Unmapped,
|
||||
Mapping,
|
||||
Mapped,
|
||||
};
|
||||
static std::atomic s_mappingState{BufferMapState::Unmapped};
|
||||
static StagingMapState s_mappingState;
|
||||
static wgpu::Limits g_cachedLimits;
|
||||
// Advanced once per logical frame in the seal prologue, under the renderer GPU mutex and with the
|
||||
// producer blocked, so every later reader sees a value that no longer moves.
|
||||
@@ -1048,7 +1044,7 @@ void initialize() {
|
||||
label.c_str());
|
||||
}
|
||||
currentStagingBuffer = 0;
|
||||
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
|
||||
s_mappingState.reset();
|
||||
map_staging_buffer();
|
||||
|
||||
{
|
||||
@@ -1160,6 +1156,8 @@ void shutdown() {
|
||||
g_uniformBuffer = {};
|
||||
g_indexBuffer = {};
|
||||
g_storageBuffer = {};
|
||||
// Invalidate outstanding callbacks before releasing their buffers.
|
||||
s_mappingState.reset();
|
||||
g_stagingBuffers.fill({});
|
||||
for (auto& pool : g_resolveSourceSnapshotPools) {
|
||||
pool.entry.reset();
|
||||
@@ -1178,27 +1176,25 @@ void shutdown() {
|
||||
g_inOffscreen = false;
|
||||
g_frameIndex = UINT32_MAX;
|
||||
currentStagingBuffer = 0;
|
||||
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
|
||||
}
|
||||
|
||||
void map_staging_buffer() {
|
||||
auto expected = BufferMapState::Unmapped;
|
||||
if (!s_mappingState.compare_exchange_strong(expected, BufferMapState::Mapping, std::memory_order_acq_rel,
|
||||
std::memory_order_acquire)) {
|
||||
const auto generation = s_mappingState.request();
|
||||
if (generation == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
g_stagingBuffers[currentStagingBuffer].MapAsync(
|
||||
wgpu::MapMode::Write, 0, StagingBufferSize, wgpu::CallbackMode::AllowSpontaneous,
|
||||
[](wgpu::MapAsyncStatus status, wgpu::StringView message) {
|
||||
[generation](wgpu::MapAsyncStatus status, wgpu::StringView message) {
|
||||
const auto result = status == wgpu::MapAsyncStatus::Success ? BufferMapState::Mapped : BufferMapState::Unmapped;
|
||||
if (!s_mappingState.complete(generation, result)) return;
|
||||
if (status == wgpu::MapAsyncStatus::CallbackCancelled || status == wgpu::MapAsyncStatus::Aborted) {
|
||||
Log.warn("Buffer mapping {}: {}", magic_enum::enum_name(status), message);
|
||||
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
|
||||
return;
|
||||
}
|
||||
ASSERT(status == wgpu::MapAsyncStatus::Success, "Buffer mapping failed: {} {}", magic_enum::enum_name(status),
|
||||
message);
|
||||
s_mappingState.store(BufferMapState::Mapped, std::memory_order_release);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1208,7 +1204,7 @@ static bool begin_frame_impl(bool clearEfb) {
|
||||
ZoneScopedN("Wait for buffer map");
|
||||
map_staging_buffer();
|
||||
while (true) {
|
||||
const auto mappingState = s_mappingState.load(std::memory_order_acquire);
|
||||
const auto mappingState = s_mappingState.state();
|
||||
if (mappingState == BufferMapState::Mapped) {
|
||||
break;
|
||||
}
|
||||
@@ -1224,6 +1220,9 @@ static bool begin_frame_impl(bool clearEfb) {
|
||||
return false;
|
||||
}
|
||||
g_instance.ProcessEvents();
|
||||
webgpu::fail_if_device_lost();
|
||||
// Sleep until the map callback lands (or 1 ms passes) instead of spinning a core on ProcessEvents.
|
||||
s_mappingState.wait_for_progress();
|
||||
}
|
||||
}
|
||||
g_recordingSnapshotSlot = currentStagingBuffer;
|
||||
@@ -1296,12 +1295,12 @@ void abort_frame() noexcept {
|
||||
g_textureUploads.clear();
|
||||
g_textureUpload.release();
|
||||
}
|
||||
if (s_mappingState.load(std::memory_order_acquire) == BufferMapState::Mapped) {
|
||||
if (s_mappingState.state() == BufferMapState::Mapped) {
|
||||
// Pending interpolation tasks hold raw pointers into the mapped staging
|
||||
// range; they must be dropped before the buffer is unmapped and rotated.
|
||||
gx::drop_pending_frame_interpolation_uniforms();
|
||||
g_stagingBuffers[currentStagingBuffer].Unmap();
|
||||
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
|
||||
s_mappingState.reset();
|
||||
currentStagingBuffer = (currentStagingBuffer + 1) % g_stagingBuffers.size();
|
||||
map_staging_buffer();
|
||||
}
|
||||
@@ -1741,7 +1740,7 @@ static bool end_batch_impl(const wgpu::CommandEncoder& cmd, bool advanceFrame,
|
||||
g_uniformUploadDestination = nullptr;
|
||||
}
|
||||
g_stagingBuffers[currentStagingBuffer].Unmap();
|
||||
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
|
||||
s_mappingState.reset();
|
||||
g_stats.drawCallCount = g_drawCallCount;
|
||||
g_stats.mergedDrawCallCount = g_mergedDrawCallCount;
|
||||
g_stats.lastVertSize = writeBuffer(g_verts, g_vertexBuffer, VertexBufferSize, "Vertex");
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
#pragma once
|
||||
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <cstdint>
|
||||
#include <mutex>
|
||||
|
||||
namespace aurora::gfx {
|
||||
|
||||
enum class BufferMapState { Unmapped, Mapping, Mapped };
|
||||
|
||||
// The renderer owns request/reset; Dawn may complete a request on another thread.
|
||||
// An old callback must never publish readiness for a different staging slot.
|
||||
class StagingMapState {
|
||||
mutable std::mutex mutex_;
|
||||
std::condition_variable changed_;
|
||||
uint64_t generation_ = 0;
|
||||
BufferMapState state_ = BufferMapState::Unmapped;
|
||||
|
||||
public:
|
||||
uint64_t request() {
|
||||
std::lock_guard lock(mutex_);
|
||||
if (state_ != BufferMapState::Unmapped) return 0;
|
||||
state_ = BufferMapState::Mapping;
|
||||
return ++generation_;
|
||||
}
|
||||
|
||||
bool complete(uint64_t generation, BufferMapState state) {
|
||||
{
|
||||
std::lock_guard lock(mutex_);
|
||||
if (generation != generation_ || state_ != BufferMapState::Mapping) return false;
|
||||
state_ = state;
|
||||
}
|
||||
changed_.notify_all();
|
||||
return true;
|
||||
}
|
||||
|
||||
void reset() {
|
||||
{
|
||||
std::lock_guard lock(mutex_);
|
||||
++generation_;
|
||||
state_ = BufferMapState::Unmapped;
|
||||
}
|
||||
changed_.notify_all();
|
||||
}
|
||||
|
||||
BufferMapState state() const {
|
||||
std::lock_guard lock(mutex_);
|
||||
return state_;
|
||||
}
|
||||
|
||||
void wait_for_progress() {
|
||||
std::unique_lock lock(mutex_);
|
||||
// ProcessEvents is still serviced between waits for implementations that
|
||||
// need it. A spontaneous completion wakes immediately, without polling.
|
||||
changed_.wait_for(lock, std::chrono::milliseconds(1), [&] { return state_ != BufferMapState::Mapping; });
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace aurora::gfx
|
||||
@@ -1,6 +1,9 @@
|
||||
#include "gx_test_common.hpp"
|
||||
#include "gfx/staging_map.hpp"
|
||||
#include "gx/pipeline.hpp"
|
||||
|
||||
#include <thread>
|
||||
|
||||
using aurora::gx::g_gxState;
|
||||
|
||||
namespace {
|
||||
@@ -84,3 +87,33 @@ TEST_F(GXFifoTest, SingleExpandedPrimitiveCannotMergeWithTriangles) {
|
||||
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
|
||||
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
|
||||
}
|
||||
|
||||
TEST(StagingMapping, RetiredCallbacksCannotPublishAnotherBuffersReadiness) {
|
||||
using namespace aurora::gfx;
|
||||
StagingMapState state;
|
||||
const auto old = state.request();
|
||||
EXPECT_EQ(state.request(), 0u);
|
||||
state.reset();
|
||||
const auto current = state.request();
|
||||
EXPECT_FALSE(state.complete(old, BufferMapState::Mapped));
|
||||
EXPECT_FALSE(state.complete(old, BufferMapState::Unmapped));
|
||||
EXPECT_EQ(state.state(), BufferMapState::Mapping);
|
||||
EXPECT_TRUE(state.complete(current, BufferMapState::Mapped));
|
||||
EXPECT_FALSE(state.complete(current, BufferMapState::Unmapped));
|
||||
EXPECT_EQ(state.state(), BufferMapState::Mapped);
|
||||
}
|
||||
|
||||
TEST(StagingMapping, AsyncCompletionWakesWaiters) {
|
||||
using namespace aurora::gfx;
|
||||
StagingMapState state;
|
||||
const auto generation = state.request();
|
||||
std::thread callback([&] {
|
||||
std::this_thread::sleep_for(std::chrono::milliseconds(10));
|
||||
state.complete(generation, BufferMapState::Mapped);
|
||||
});
|
||||
const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(2);
|
||||
while (state.state() == BufferMapState::Mapping && std::chrono::steady_clock::now() < deadline)
|
||||
state.wait_for_progress();
|
||||
callback.join();
|
||||
EXPECT_EQ(state.state(), BufferMapState::Mapped);
|
||||
}
|
||||
Reference in new issue
Block a user