Merge pull request #8 from mitch030504/claude/project-thread-by97b7

Port upstream renderer fixes and stop spinning on staging buffer maps
This commit is contained in:
mitch030504 authored and GitHub committed 2026-10-05 15:48:16 +02:00
commit f2f4f081c8
19 files changed
+467 -65

No files matched your search

+8
View File
@@ -295,12 +295,20 @@ typedef struct {
bool xrHeadsetOnly;
} AuroraConfig;
typedef enum {
AURORA_INITIALIZATION_SUCCESS = 0,
AURORA_INITIALIZATION_GRAPHICS_UNAVAILABLE = 1,
} AuroraInitializationStatus;
typedef struct {
AuroraBackend backend;
const char* userPath;
const char* cachePath;
SDL_Window* window;
AuroraWindowSize windowSize;
AuroraInitializationStatus initializationStatus;
// On failure, owned by SDL on the calling thread. Copy before another SDL call.
const char* initializationError;
} AuroraInfo;
AuroraInfo aurora_initialize(int argc, char* argv[], const AuroraConfig* config);
+19 -1
View File
@@ -1229,15 +1229,23 @@ AuroraInfo initialize(int argc, char* argv[], const AuroraConfig& config) noexce
const AuroraBackend requestedBackend = config.desiredBackend;
AuroraBackend selectedBackend = requestedBackend;
bool windowCreated = false;
std::string firstGraphicsError;
const auto rememberGraphicsError = [&] {
if (firstGraphicsError.empty() && SDL_GetError()[0] != '\0') {
firstGraphicsError = SDL_GetError();
}
};
if (selectedBackend != BACKEND_AUTO) {
Log.info("Requested graphics backend: {}", backend_name(selectedBackend));
if (window::create_window(selectedBackend)) {
if (webgpu::initialize(selectedBackend)) {
windowCreated = true;
} else {
rememberGraphicsError();
window::destroy_window();
}
} else {
rememberGraphicsError();
Log.error("Failed to create a window for backend {}: {}", backend_name(selectedBackend), SDL_GetError());
}
if (!windowCreated) {
@@ -1254,18 +1262,28 @@ AuroraInfo initialize(int argc, char* argv[], const AuroraConfig& config) noexce
for (const auto backendType : PreferredBackendOrder) {
selectedBackend = backendType;
if (!window::create_window(selectedBackend)) {
rememberGraphicsError();
continue;
}
if (webgpu::initialize(selectedBackend)) {
windowCreated = true;
break;
} else {
rememberGraphicsError();
window::destroy_window();
}
}
}
ASSERT(windowCreated, "Error creating window: {}", SDL_GetError());
if (!windowCreated) {
if (firstGraphicsError.empty()) firstGraphicsError = "No supported graphics backend is available";
SDL_SetError("%s", firstGraphicsError.c_str());
Log.error("Graphics initialization failed: {}", firstGraphicsError);
return {
.initializationStatus = AURORA_INITIALIZATION_GRAPHICS_UNAVAILABLE,
.initializationError = SDL_GetError(),
};
}
if (requestedBackend != BACKEND_AUTO && selectedBackend != requestedBackend) {
Log.error(
"Graphics backend fallback in effect: video.graphics_api requested {}, "
+27 -12
View File
@@ -7,11 +7,13 @@
#include <algorithm>
#include <atomic>
#include <optional>
#include <mutex>
namespace aurora::vi {
std::optional<GXRenderModeObj> g_renderMode;
namespace {
std::atomic<float> g_presentAspectCorrection{1.f};
std::mutex g_renderModeMutex;
float calculate_present_aspect_correction(const GXRenderModeObj& rm) noexcept {
if (rm.viWidth == 0 || rm.viHeight == 0) {
@@ -29,9 +31,8 @@ float calculate_present_aspect_correction(const GXRenderModeObj& rm) noexcept {
const float verticalFill = static_cast<float>(rm.viHeight) / nominalActiveHeight;
return horizontalFill / verticalFill;
}
} // namespace
Vec2<uint32_t> render_mode_size() noexcept {
Vec2<uint32_t> render_mode_size_locked() noexcept {
if (!g_renderMode) {
return {640, 528};
}
@@ -40,18 +41,31 @@ Vec2<uint32_t> render_mode_size() noexcept {
return {std::max<uint32_t>(g_renderMode->fbWidth, 640), std::max<uint32_t>(g_renderMode->efbHeight, 528)};
}
} // namespace
Vec2<uint32_t> render_mode_size() noexcept {
std::lock_guard lock(g_renderModeMutex);
return render_mode_size_locked();
}
void configure(const GXRenderModeObj* rm) noexcept {
const auto oldSize = render_mode_size();
if (rm == nullptr) {
g_renderMode.reset();
} else {
g_renderMode = *rm;
g_presentAspectCorrection.store(calculate_present_aspect_correction(*rm), std::memory_order_release);
bool sizeChanged = false;
{
std::lock_guard lock(g_renderModeMutex);
const auto oldSize = render_mode_size_locked();
if (rm == nullptr) {
g_renderMode.reset();
} else {
g_renderMode = *rm;
g_presentAspectCorrection.store(calculate_present_aspect_correction(*rm), std::memory_order_release);
}
if (rm == nullptr) {
g_presentAspectCorrection.store(1.f, std::memory_order_release);
}
sizeChanged = render_mode_size_locked() != oldSize;
}
if (rm == nullptr) {
g_presentAspectCorrection.store(1.f, std::memory_order_release);
}
if (render_mode_size() != oldSize) {
// Never hold the mode lock across a resize request or a renderer callback.
if (sizeChanged) {
window::request_frame_buffer_resize();
}
}
@@ -61,6 +75,7 @@ Vec2<uint32_t> configured_fb_size() noexcept {
}
Vec2<uint32_t> visible_fb_size() noexcept {
std::lock_guard lock(g_renderModeMutex);
if (!g_renderMode) {
return {640, 528};
}
+18 -19
View File
@@ -1,4 +1,5 @@
#include "common.hpp"
#include "staging_map.hpp"
#include "../gx/shader_info.hpp"
#include "clear.hpp"
@@ -139,12 +140,7 @@ wgpu::Buffer g_storageBuffer;
constexpr size_t FrameSlotCount = 3;
static std::array<wgpu::Buffer, FrameSlotCount> g_stagingBuffers;
static size_t currentStagingBuffer = 0;
enum class BufferMapState {
Unmapped,
Mapping,
Mapped,
};
static std::atomic s_mappingState{BufferMapState::Unmapped};
static StagingMapState s_mappingState;
static wgpu::Limits g_cachedLimits;
// Advanced once per logical frame in the seal prologue, under the renderer GPU mutex and with the
// producer blocked, so every later reader sees a value that no longer moves.
@@ -405,7 +401,7 @@ static size_t g_recordingSnapshotSlot = 0;
static TextureHandle new_resolve_source_snapshot(wgpu::Extent3D size, wgpu::TextureFormat format) noexcept {
const wgpu::TextureDescriptor textureDescriptor{
.label = "GX Copy Source Snapshot",
.usage = wgpu::TextureUsage::TextureBinding | wgpu::TextureUsage::CopyDst,
.usage = wgpu::TextureUsage::TextureBinding | wgpu::TextureUsage::CopySrc | wgpu::TextureUsage::CopyDst,
.dimension = wgpu::TextureDimension::e2D,
.size = size,
.format = format,
@@ -1048,7 +1044,7 @@ void initialize() {
label.c_str());
}
currentStagingBuffer = 0;
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
s_mappingState.reset();
map_staging_buffer();
{
@@ -1160,6 +1156,8 @@ void shutdown() {
g_uniformBuffer = {};
g_indexBuffer = {};
g_storageBuffer = {};
// Invalidate outstanding callbacks before releasing their buffers.
s_mappingState.reset();
g_stagingBuffers.fill({});
for (auto& pool : g_resolveSourceSnapshotPools) {
pool.entry.reset();
@@ -1178,27 +1176,25 @@ void shutdown() {
g_inOffscreen = false;
g_frameIndex = UINT32_MAX;
currentStagingBuffer = 0;
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
}
void map_staging_buffer() {
auto expected = BufferMapState::Unmapped;
if (!s_mappingState.compare_exchange_strong(expected, BufferMapState::Mapping, std::memory_order_acq_rel,
std::memory_order_acquire)) {
const auto generation = s_mappingState.request();
if (generation == 0) {
return;
}
g_stagingBuffers[currentStagingBuffer].MapAsync(
wgpu::MapMode::Write, 0, StagingBufferSize, wgpu::CallbackMode::AllowSpontaneous,
[](wgpu::MapAsyncStatus status, wgpu::StringView message) {
[generation](wgpu::MapAsyncStatus status, wgpu::StringView message) {
const auto result = status == wgpu::MapAsyncStatus::Success ? BufferMapState::Mapped : BufferMapState::Unmapped;
if (!s_mappingState.complete(generation, result)) return;
if (status == wgpu::MapAsyncStatus::CallbackCancelled || status == wgpu::MapAsyncStatus::Aborted) {
Log.warn("Buffer mapping {}: {}", magic_enum::enum_name(status), message);
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
return;
}
ASSERT(status == wgpu::MapAsyncStatus::Success, "Buffer mapping failed: {} {}", magic_enum::enum_name(status),
message);
s_mappingState.store(BufferMapState::Mapped, std::memory_order_release);
});
}
@@ -1208,7 +1204,7 @@ static bool begin_frame_impl(bool clearEfb) {
ZoneScopedN("Wait for buffer map");
map_staging_buffer();
while (true) {
const auto mappingState = s_mappingState.load(std::memory_order_acquire);
const auto mappingState = s_mappingState.state();
if (mappingState == BufferMapState::Mapped) {
break;
}
@@ -1224,6 +1220,9 @@ static bool begin_frame_impl(bool clearEfb) {
return false;
}
g_instance.ProcessEvents();
webgpu::fail_if_device_lost();
// Sleep until the map callback lands (or 1 ms passes) instead of spinning a core on ProcessEvents.
s_mappingState.wait_for_progress();
}
}
g_recordingSnapshotSlot = currentStagingBuffer;
@@ -1296,12 +1295,12 @@ void abort_frame() noexcept {
g_textureUploads.clear();
g_textureUpload.release();
}
if (s_mappingState.load(std::memory_order_acquire) == BufferMapState::Mapped) {
if (s_mappingState.state() == BufferMapState::Mapped) {
// Pending interpolation tasks hold raw pointers into the mapped staging
// range; they must be dropped before the buffer is unmapped and rotated.
gx::drop_pending_frame_interpolation_uniforms();
g_stagingBuffers[currentStagingBuffer].Unmap();
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
s_mappingState.reset();
currentStagingBuffer = (currentStagingBuffer + 1) % g_stagingBuffers.size();
map_staging_buffer();
}
@@ -1741,7 +1740,7 @@ static bool end_batch_impl(const wgpu::CommandEncoder& cmd, bool advanceFrame,
g_uniformUploadDestination = nullptr;
}
g_stagingBuffers[currentStagingBuffer].Unmap();
s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release);
s_mappingState.reset();
g_stats.drawCallCount = g_drawCallCount;
g_stats.mergedDrawCallCount = g_mergedDrawCallCount;
g_stats.lastVertSize = writeBuffer(g_verts, g_vertexBuffer, VertexBufferSize, "Vertex");
+2 -1
View File
@@ -109,8 +109,9 @@ void ensure_native_texture(PendingCopy& pending, TextureHandle* cache = nullptr)
*cache = pending.nativeTexture;
}
}
// The shared blit shader clamps Y to flags.z/w; preserve the full source.
const std::array nativeBlitUniform{
0.0f, 0.0f, 1.0f, 1.0f, 0.0f, 64.0f, 0.0f, 0.0f, 0.0f, 1.0f, 0.0f, 0.0f,
0.0f, 0.0f, 1.0f, 1.0f, 0.0f, 64.0f, 0.0f, 0.0f, 0.0f, 1.0f, 0.0f, 1.0f,
};
pending.nativeBlitUniform = push_uniform(nativeBlitUniform);
}
+3 -1
View File
@@ -426,6 +426,7 @@ static PendingPipeline* touch_pending_pipeline(PipelineRef hash, bool prioritize
g_priorityPipelines.emplace_back(std::move(*backgroundIt));
g_backgroundPipelines.erase(backgroundIt);
g_pipelineCv.notify_all();
return &g_priorityPipelines.back();
}
@@ -566,7 +567,8 @@ static PipelineRef find_pipeline_impl(ShaderType type, const PipelineConfig& con
}
if (notifyWorker) {
g_pipelineCv.notify_one();
// Compiler workers and renderer waiters share this condition variable.
g_pipelineCv.notify_all();
}
if (notifyWaiters) {
g_pipelineCv.notify_all();
+60
View File
@@ -0,0 +1,60 @@
#pragma once
#include <chrono>
#include <condition_variable>
#include <cstdint>
#include <mutex>
namespace aurora::gfx {
enum class BufferMapState { Unmapped, Mapping, Mapped };
// The renderer owns request/reset; Dawn may complete a request on another thread.
// An old callback must never publish readiness for a different staging slot.
class StagingMapState {
mutable std::mutex mutex_;
std::condition_variable changed_;
uint64_t generation_ = 0;
BufferMapState state_ = BufferMapState::Unmapped;
public:
uint64_t request() {
std::lock_guard lock(mutex_);
if (state_ != BufferMapState::Unmapped) return 0;
state_ = BufferMapState::Mapping;
return ++generation_;
}
bool complete(uint64_t generation, BufferMapState state) {
{
std::lock_guard lock(mutex_);
if (generation != generation_ || state_ != BufferMapState::Mapping) return false;
state_ = state;
}
changed_.notify_all();
return true;
}
void reset() {
{
std::lock_guard lock(mutex_);
++generation_;
state_ = BufferMapState::Unmapped;
}
changed_.notify_all();
}
BufferMapState state() const {
std::lock_guard lock(mutex_);
return state_;
}
void wait_for_progress() {
std::unique_lock lock(mutex_);
// ProcessEvents is still serviced between waits for implementations that
// need it. A spontaneous completion wakes immediately, without polling.
changed_.wait_for(lock, std::chrono::milliseconds(1), [&] { return state_ != BufferMapState::Mapping; });
}
};
} // namespace aurora::gfx
+20 -6
View File
@@ -422,7 +422,7 @@ static wgpu::BindGroupLayout g_depthBindGroupLayout;
static wgpu::Sampler g_nearestSampler;
static wgpu::Sampler g_linearSampler;
static absl::flat_hash_map<GXTexFmt, wgpu::RenderPipeline> g_pipelines;
static wgpu::RenderPipeline g_blitPipeline;
static absl::flat_hash_map<wgpu::TextureFormat, wgpu::RenderPipeline> g_blitPipelines;
static wgpu::RenderPipeline g_quest1BlitPipeline;
static bool quest1_simple_blit() noexcept {
@@ -549,9 +549,15 @@ void initialize() {
};
g_depthBindGroupLayout = g_device.CreateBindGroupLayout(&depthBindGroupLayoutDescriptor);
g_blitPipeline = create_pipeline(
{GX_TF_RGBA8, FragPassthrough, webgpu::g_graphicsConfig.surfaceConfiguration.format, "TexCopyConv Blit"},
ShaderPreamble, g_bindGroupLayout);
// Native RAM readback uses RGBA even when the EFB/surface uses BGRA.
// Build both variants here; frame workers only read the completed map.
for (const auto format : {wgpu::TextureFormat::RGBA8Unorm, wgpu::TextureFormat::BGRA8Unorm,
webgpu::g_graphicsConfig.surfaceConfiguration.format}) {
if (format != wgpu::TextureFormat::Undefined && !g_blitPipelines.contains(format)) {
g_blitPipelines[format] = create_pipeline({GX_TF_RGBA8, FragPassthrough, format, "TexCopyConv Blit"},
ShaderPreamble, g_bindGroupLayout);
}
}
g_quest1BlitPipeline = create_pipeline(
{GX_TF_RGBA8, {}, webgpu::g_graphicsConfig.surfaceConfiguration.format, "Quest 1 Simple TexCopy Blit"},
SimpleBlitShader, g_bindGroupLayout);
@@ -585,7 +591,7 @@ void initialize() {
void shutdown() {
g_pipelines.clear();
g_blitPipeline = {};
g_blitPipelines.clear();
g_quest1BlitPipeline = {};
g_bindGroupLayout = {};
g_depthBindGroupLayout = {};
@@ -678,7 +684,15 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
}
void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
execute(cmd, req, quest1_simple_blit() ? g_quest1BlitPipeline : g_blitPipeline);
if (quest1_simple_blit()) {
execute(cmd, req, g_quest1BlitPipeline);
return;
}
const auto it = g_blitPipelines.find(req.dst->format);
if (it == g_blitPipelines.end()) {
Log.fatal("Unsupported blit destination format {}", static_cast<int>(req.dst->format));
}
execute(cmd, req, it->second);
}
} // namespace aurora::gfx::tex_copy_conv
+40 -12
View File
@@ -33,10 +33,11 @@ using IndexBuffer = std::vector<u16>;
static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount) {
size_t writePos = 0;
if (prim == GX_QUADS) {
// Retain the existing incomplete-quad behavior: every started group emits a complete six-index quad.
buf.resize(((static_cast<u32>(vtxCount) + 3u) / 4u) * 6u);
// GX renders a three-vertex remainder as a triangle. One/two are ignored.
const u32 completeVertices = static_cast<u32>(vtxCount) & ~3u;
buf.resize((completeVertices / 4u) * 6u + (vtxCount % 4u == 3u ? 3u : 0u));
for (u16 v = 0; v < vtxCount; v += 4) {
for (u32 v = 0; v < completeVertices; v += 4) {
const u16 idx0 = v;
const u16 idx1 = static_cast<u16>(v + 1);
const u16 idx2 = static_cast<u16>(v + 2);
@@ -48,15 +49,21 @@ static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount
buf[writePos++] = idx3;
buf[writePos++] = idx0;
}
if (vtxCount % 4u == 3u) {
buf[writePos++] = static_cast<u16>(completeVertices);
buf[writePos++] = static_cast<u16>(completeVertices + 1u);
buf[writePos++] = static_cast<u16>(completeVertices + 2u);
}
} else if (prim == GX_TRIANGLES) {
buf.resize(vtxCount);
for (u16 v = 0; v < vtxCount; ++v) {
const u32 completeVertices = (static_cast<u32>(vtxCount) / 3u) * 3u;
buf.resize(completeVertices);
for (u32 v = 0; v < completeVertices; ++v) {
buf[writePos++] = v;
}
} else if (prim == GX_TRIANGLEFAN) {
const u32 indexCount = vtxCount <= 3 ? vtxCount : 3u + (static_cast<u32>(vtxCount) - 3u) * 3u;
const u32 indexCount = vtxCount < 3 ? 0u : (static_cast<u32>(vtxCount) - 2u) * 3u;
buf.resize(indexCount);
for (u16 v = 0; v < vtxCount; ++v) {
for (u32 v = 0; indexCount != 0 && v < vtxCount; ++v) {
if (v < 3) {
buf[writePos++] = v;
continue;
@@ -66,9 +73,9 @@ static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount
buf[writePos++] = v;
}
} else if (prim == GX_TRIANGLESTRIP) {
const u32 indexCount = vtxCount <= 3 ? vtxCount : 3u + (static_cast<u32>(vtxCount) - 3u) * 3u;
const u32 indexCount = vtxCount < 3 ? 0u : (static_cast<u32>(vtxCount) - 2u) * 3u;
buf.resize(indexCount);
for (u16 v = 0; v < vtxCount; ++v) {
for (u32 v = 0; indexCount != 0 && v < vtxCount; ++v) {
if (v < 3) {
buf[writePos++] = v;
continue;
@@ -91,6 +98,13 @@ static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount
return static_cast<u32>(writePos);
}
// Empty/incomplete draws consume FIFO bytes but cannot produce a primitive.
static bool has_complete_primitive(GXPrimitive prim, u16 count) {
if (prim == GX_POINTS) return count >= 1;
if (prim == GX_LINES || prim == GX_LINESTRIP) return count >= 2;
return count >= 3;
}
// GX FIFO opcodes - use CP_ prefix to avoid clashing with GXCommandList.h macros
static constexpr u8 CP_CMD_NOP = GX_NOP;
static constexpr u8 CP_CMD_LOAD_CP_REG = GX_LOAD_CP_REG;
@@ -552,6 +566,10 @@ void process(const u8* data, u32 size, bool bigEndian) {
for (int i = GX_VA_POS; i <= GX_VA_TEX7; ++i) {
g_gxState.arrays[i].cachedRange = {};
}
// A merged draw retains its previous array uploads. Force a new draw so
// handle_draw_unmerged observes the invalidation and uploads fresh data.
// Pipeline configuration itself did not change.
g_gxState.stateDirty = true;
break;
}
@@ -1926,6 +1944,10 @@ static u32 calculate_last_vtx_size(GXVtxFmt fmt) {
g_gxState.lastVtxFmt = fmt;
g_gxState.lastVtxSize = vtxSize;
// The format is selected by the draw opcode, without a register write.
// Even equal-stride formats may decode bytes differently, so do not merge
// into a draw using the previous format's shader and uniform layout.
g_gxState.stateDirty = true;
return vtxSize;
}
@@ -2366,6 +2388,8 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui
return false;
}
if (!has_complete_primitive(prim, vtxCount)) return true;
// This entry point bypasses process(), so it owns the renderer lock itself.
std::lock_guard gpuLock(aurora::renderer_gpu_mutex());
if (model_array_hidden()) return true;
@@ -2414,7 +2438,7 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
return false;
}
if (model_array_hidden()) {
if (!has_complete_primitive(prim, vtxCount) || model_array_hidden()) {
pos += totalVtxBytes;
return true;
}
@@ -2436,9 +2460,12 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
// Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw
// that resolved the same animated array: the merged whole renders through that draw's binding. Anything the
// decision cache cannot vouch for (a command it was not recorded against) stays unmerged.
// Expanded lines/points have different vertex interpretation even with one instance.
// Triangle-list output has no restart index; index 65535 is usable.
// Overflow would address earlier vertices instead of the appended geometry.
if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS &&
!lastDraw->uniformReplayLayout.vertexMotion.enabled &&
lastDraw->instanceCount == 1 &&
!lastDraw->uniformReplayLayout.vertexMotion.enabled && !lastDraw->expandedPrimitive &&
lastDraw->instanceCount == 1 && uint64_t(lastDraw->vtxCount) + vtxCount <= 65536u &&
(nativeWheelArrays.empty() ||
(nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel)))
LIKELY {
@@ -2615,6 +2642,7 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
.vtxCount = vtxCount,
.indexCount = numIndices,
.instanceCount = instanceCount,
.expandedPrimitive = prim == GX_LINES || prim == GX_LINESTRIP || prim == GX_POINTS,
.bindGroups = bindGroups,
.dstAlpha = pipelineState.dstAlpha,
.screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride),
+2
View File
@@ -439,6 +439,8 @@ struct GXState {
u32 pipelineStateGeneration = next_gx_state_epoch();
std::array<u32, 0x100> bpRegCache = [] {
std::array<u32, 0x100> regs{};
// Force the first GEN_MODE decode without changing its masked reset value.
regs[0x00] = 0xFF000000;
regs[0xFE] = 0x00FFFFFF;
return regs;
}();
+1
View File
@@ -29,6 +29,7 @@ struct DrawData {
uint32_t vtxCount;
uint32_t indexCount;
uint32_t instanceCount;
bool expandedPrimitive;
GXBindGroups bindGroups;
uint32_t dstAlpha;
// Valid only for simple orthographic rectangles/lines (textured or not).
+16 -2
View File
@@ -1739,8 +1739,22 @@ fn load_u16(p: ptr<storage, array<u32>>, byte_off: u32, le: bool) -> u32 {{
return bswap16(raw, le);
}}
fn load_u24_raw(p: ptr<storage, array<u32>>, byte_off: u32) -> u32 {{
let word_idx = byte_off >> 2u;
let sub = byte_off & 3u;
let word = p[word_idx];
// Three bytes at offsets zero or one fit entirely in this word. Do not
// access the next word: this attribute may end at the binding boundary.
if (sub <= 1u) {{
return (word >> (sub * 8u)) & 0x00FFFFFFu;
}}
let next = p[word_idx + 1u];
let shift = sub * 8u;
return ((word >> shift) | (next << (32u - shift))) & 0x00FFFFFFu;
}}
fn load_u24(p: ptr<storage, array<u32>>, byte_off: u32, le: bool) -> u32 {{
let raw = load_u32_raw(p, byte_off) & 0x00FFFFFFu;
let raw = load_u24_raw(p, byte_off);
if (le) {{
return raw;
}}
@@ -1780,7 +1794,7 @@ fn raw_fetch_u8_2(p: ptr<storage, array<u32>>, byte_off: u32) -> vec2u {{
}}
fn raw_fetch_u8_3(p: ptr<storage, array<u32>>, byte_off: u32) -> vec3u {{
let raw = load_u32_raw(p, byte_off);
let raw = load_u24_raw(p, byte_off);
return vec3u(
extractBits(raw, 0u, 8u),
extractBits(raw, 8u, 8u),
+18 -6
View File
@@ -85,18 +85,30 @@ void initialize() noexcept {
void shutdown() noexcept {
ZoneScoped;
if (g_useSdlRenderer) {
ImGui_ImplSDLRenderer3_Shutdown();
} else {
ImGui_ImplWGPU_Shutdown();
// Startup can fail before either backend initializes. A context alone does
// not mean its renderer/platform backend owns resources to release.
if (ImGui::GetCurrentContext() != nullptr) {
ImGuiIO& io = ImGui::GetIO();
if (io.BackendRendererUserData != nullptr) {
if (g_useSdlRenderer) {
ImGui_ImplSDLRenderer3_Shutdown();
} else {
ImGui_ImplWGPU_Shutdown();
}
}
if (io.BackendPlatformUserData != nullptr) {
ImGui_ImplSDL3_Shutdown();
}
ImGui::DestroyContext();
}
ImGui_ImplSDL3_Shutdown();
ImGui::DestroyContext();
for (const auto& texture : g_sdlTextures) {
SDL_DestroyTexture(texture);
}
g_sdlTextures.clear();
g_wgpuTextures.clear();
g_useSdlRenderer = false;
g_scale = 0.f;
g_frameDataBuilt = true;
}
void process_event(const SDL_Event& event) noexcept {
+9
View File
@@ -627,12 +627,16 @@ bool initialize(AuroraBackend auroraBackend) {
g_adapter = std::move(adapter);
} else {
Log.warn("Adapter request failed: {}", message);
const std::string_view reason{message};
SDL_SetError("Graphics adapter unavailable: %.*s",
static_cast<int>(std::min<size_t>(reason.size(), 512)), reason.data());
}
});
const auto status = g_instance.WaitAny(future, 5000000000);
if (status != wgpu::WaitStatus::Success) {
Log.error("Failed to create {} adapter: {}", magic_enum::enum_name(backend),
magic_enum::enum_name(status));
SDL_SetError("Graphics adapter request did not complete within its startup deadline");
return false;
}
if (!g_adapter) {
@@ -845,11 +849,16 @@ bool initialize(AuroraBackend auroraBackend) {
g_device = std::move(device);
} else {
Log.warn("Device request failed: {}", message);
const std::string_view reason{message};
SDL_SetError("Graphics device unavailable: %.*s",
static_cast<int>(std::min<size_t>(reason.size(), 512)),
reason.data());
}
});
const auto status = g_instance.WaitAny(future, 5000000000);
if (status != wgpu::WaitStatus::Success) {
Log.error("Failed to create device: {}", magic_enum::enum_name(status));
SDL_SetError("Graphics device request did not complete within its startup deadline");
return false;
}
if (!g_device) {
+28 -2
View File
@@ -6,6 +6,7 @@
#include <deque>
#include <memory>
#include <mutex>
#include <limits>
#include <string>
#include <filesystem>
#include <thread>
@@ -313,8 +314,33 @@ static size_t load_from_database(const XXH128_hash_t& keyHash, void* value, size
if (ret == SQLITE_ROW) {
// Hit
const auto foundPtr = sqlite3_column_blob(load_stmt, 0);
foundSize = sqlite3_column_int64(load_stmt, 1);
const bool compressed = sqlite3_column_int(load_stmt, 2) != 0;
const auto declaredSize = sqlite3_column_int64(load_stmt, 1);
const auto storedSize = sqlite3_column_bytes(load_stmt, 0);
const auto compression = sqlite3_column_int(load_stmt, 2);
const bool compressed = compression == 1;
// Dawn asks for the size before allocating its destination. Validate here,
// not only during the copy: corrupt metadata must become a cache miss.
bool valid = declaredSize > 0 &&
static_cast<uint64_t>(declaredSize) <= std::numeric_limits<size_t>::max() &&
foundPtr != nullptr && storedSize > 0 && (compression == 0 || compression == 1);
if (valid && compressed) {
#if defined(AURORA_CACHE_USE_ZSTD)
// Our writer uses ZSTD_compress, which records the original content size.
const auto frameSize = ZSTD_getFrameContentSize(foundPtr, static_cast<size_t>(storedSize));
valid = frameSize != ZSTD_CONTENTSIZE_ERROR && frameSize != ZSTD_CONTENTSIZE_UNKNOWN &&
frameSize == static_cast<uint64_t>(declaredSize);
#else
valid = false;
#endif
} else if (valid) {
valid = declaredSize == storedSize;
}
if (!valid) {
Log.error("Ignoring cache entry with inconsistent size or compression metadata");
check(sqlite3_reset(load_stmt));
return 0;
}
foundSize = static_cast<size_t>(declaredSize);
if (value == nullptr) {
g_hits.fetch_add(1, std::memory_order_relaxed);
} else {
+1
View File
@@ -64,6 +64,7 @@ if (AURORA_ENABLE_GX)
add_executable(gx_fifo_tests
gx_fifo_test.cpp
gx_test_stubs.cpp
renderer_regression_test.cpp
stereo_replay_test.cpp
stereo_interpolation_test.cpp
stereo_mirror_test.cpp
+72 -3
View File
@@ -1328,12 +1328,35 @@ TEST(TevRegisterLivenessContract, PacksOneUniformWhenBothHalvesNeedInitialValue)
auto config = baseline;
config.tevStages[0].colorPass.a = GX_CC_C0;
config.tevStages[0].alphaPass.a = GX_CA_A0;
config.tevStages[0].colorPass.b = GX_CC_KONST;
config.tevStages[0].kcSel = GX_TEV_KCSEL_K0;
const auto baselineInfo = aurora::gx::build_shader_info(baseline);
const auto info = aurora::gx::build_shader_info(config);
EXPECT_TRUE(info.loadsTevRegRgb.test(GX_TEVREG0));
EXPECT_TRUE(info.loadsTevRegAlpha.test(GX_TEVREG0));
EXPECT_EQ(info.uniformSize, baselineInfo.uniformSize + sizeof(aurora::Vec4<float>));
// The final allocation is alignment-rounded, so adding one register need
// not increase it. Verify actual packing with a distinct following K color.
const auto savedReg = g_gxState.colorRegs[GX_TEVREG0];
const auto savedKColor = g_gxState.kcolors[GX_KCOLOR0];
g_gxState.colorRegs[GX_TEVREG0] = {11.f, 22.f, 33.f, 44.f};
g_gxState.kcolors[GX_KCOLOR0] = {55.f, 66.f, 77.f, 88.f};
EXPECT_TRUE(info.sampledKColors.test(GX_KCOLOR0));
aurora::gfx::testing::reset_uniform_allocations();
aurora::gx::build_uniform(info, 0, {}, {}, false);
const auto expectedReg = g_gxState.colorRegs[GX_TEVREG0];
const auto expectedKColor = g_gxState.kcolors[GX_KCOLOR0];
g_gxState.colorRegs[GX_TEVREG0] = savedReg;
g_gxState.kcolors[GX_KCOLOR0] = savedKColor;
const auto& bytes = aurora::gfx::testing::uniform_allocation(0);
const auto* reg = reinterpret_cast<const uint8_t*>(&expectedReg);
const auto found = std::search(bytes.begin(), bytes.end(), reg, reg + sizeof(aurora::Vec4<float>));
ASSERT_NE(found, bytes.end());
const size_t offset = static_cast<size_t>(found - bytes.begin());
ASSERT_LE(offset + 2 * sizeof(aurora::Vec4<float>), bytes.size());
EXPECT_EQ(std::memcmp(bytes.data() + offset + sizeof(aurora::Vec4<float>), &expectedKColor,
sizeof(aurora::Vec4<float>)),
0);
aurora::gfx::testing::reset_uniform_allocations();
}
// BP registers (direct FIFO writes, no dirty state flush needed)
@@ -1390,6 +1413,51 @@ TEST_F(GXFifoTest, BlendMode_Logic) {
EXPECT_EQ(g_gxState.blendOp, GX_LO_XOR);
}
TEST_F(GXFifoTest, GenMode_FirstZeroWriteDecodesAndRepeatDeduplicates) {
reset_gx_state();
const auto before = g_gxState.pipelineStateGeneration;
decode_fifo(bp_cmd(0, 0));
EXPECT_EQ(g_gxState.numTevStages, 1u);
EXPECT_EQ(g_gxState.cullMode, GX_CULL_NONE);
EXPECT_EQ(g_gxState.numChans, 0u);
EXPECT_EQ(g_gxState.numTexGens, 0u);
EXPECT_EQ(g_gxState.numIndStages, 0u);
EXPECT_EQ(g_gxState.bpRegCache[0], 0u);
EXPECT_NE(g_gxState.pipelineStateGeneration, before);
const auto decoded = g_gxState.pipelineStateGeneration;
decode_fifo(bp_cmd(0, 0));
EXPECT_EQ(g_gxState.pipelineStateGeneration, decoded);
}
TEST_F(GXFifoTest, GenMode_FirstMaskedWritePreservesZeroResetBits) {
for (const u32 mask : {0u, 1u << 10}) {
reset_gx_state();
const auto before = g_gxState.pipelineStateGeneration;
decode_fifo(bp_cmd(0xFE, mask));
decode_fifo(bp_cmd(0, 0xFFFFFF));
EXPECT_EQ(g_gxState.bpRegCache[0], mask);
EXPECT_EQ(g_gxState.bpRegCache[0xFE], 0xFFFFFFu);
EXPECT_EQ(g_gxState.numTevStages, mask ? 2u : 1u);
EXPECT_EQ(g_gxState.cullMode, GX_CULL_NONE);
EXPECT_NE(g_gxState.pipelineStateGeneration, before);
decode_fifo(bp_cmd(0, 0));
EXPECT_EQ(g_gxState.numTevStages, 1u);
EXPECT_EQ(g_gxState.bpRegCache[0], 0u);
}
}
TEST_F(GXFifoTest, GenMode_ColdSingleStageApiSetupDecodes) {
reset_gx_state();
GXSetNumTevStages(1);
GXSetNumTexGens(0);
GXSetNumChans(0);
GXSetCullMode(GX_CULL_NONE);
const auto bytes = flush_and_capture();
decode_fifo(bytes);
EXPECT_EQ(g_gxState.numTevStages, 1u);
EXPECT_EQ(g_gxState.cullMode, GX_CULL_NONE);
}
TEST_F(GXFifoTest, BpMask_AppliesOnlyToNextWrite) {
std::vector<u8> bytes;
auto mask = bp_cmd(0xFE, 1u << 19);
@@ -2861,6 +2929,7 @@ TEST_F(GXFifoTest, DrawTopologyTemplatesPreserveExactGxIndexOrder) {
const auto decodeAndReadIndices = [&](GXPrimitive primitive, u16 count) {
std::vector<u8> fifo;
append_test_draw(fifo, primitive, count);
aurora::gfx::testing::reset_vertex_push_record();
decode_fifo(fifo);
return aurora::gfx::testing::last_pushed_indices();
};
@@ -2871,7 +2940,7 @@ TEST_F(GXFifoTest, DrawTopologyTemplatesPreserveExactGxIndexOrder) {
g_gxState.stateDirty = true;
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 5), (std::vector<u16>{0, 1, 2, 0, 2, 3, 0, 3, 4}));
g_gxState.stateDirty = true;
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 2), (std::vector<u16>{0, 1}));
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 2), (std::vector<u16>{}));
g_gxState.stateDirty = true;
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLESTRIP, 6), (std::vector<u16>{0, 1, 2, 2, 1, 3, 2, 3, 4, 4, 3, 5}));
g_gxState.stateDirty = true;
@@ -0,0 +1,119 @@
#include "gx_test_common.hpp"
#include "gfx/staging_map.hpp"
#include "gx/pipeline.hpp"
#include <thread>
using aurora::gx::g_gxState;
namespace {
std::vector<u8> draw(GXPrimitive primitive, u16 count, GXVtxFmt format = GX_VTXFMT0) {
std::vector<u8> bytes{static_cast<u8>(primitive | format), static_cast<u8>(count >> 8), static_cast<u8>(count)};
bytes.resize(3 + count);
return bytes;
}
} // namespace
TEST_F(GXFifoTest, MaximumQuadCountTerminatesWithoutOutOfRangeIndices) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
for (const u16 count : {65532, 65533, 65534, 65535}) {
g_gxState.stateDirty = true;
decode_fifo(draw(GX_QUADS, count));
const auto& indices = aurora::gfx::testing::last_pushed_indices();
ASSERT_EQ(indices.size(), (count / 4) * 6 + (count % 4 == 3 ? 3 : 0));
for (const auto index : indices)
ASSERT_LT(index, count);
}
}
TEST_F(GXFifoTest, IncompletePrimitivesNeverJoinAcrossDraws) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_TRIANGLES, 4));
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
decode_fifo(draw(GX_TRIANGLES, 5));
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{4, 5, 6}));
const auto before = aurora::gfx::testing::last_pushed_indices();
decode_fifo(draw(GX_TRIANGLEFAN, 2));
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), before);
}
TEST_F(GXFifoTest, MergeStopsBeforeSixteenBitIndexOverflow) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_TRIANGLES, 65535));
decode_fifo(draw(GX_TRIANGLES, 3));
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
}
TEST_F(GXFifoTest, VertexCacheInvalidationBreaksDrawMerging) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_TRIANGLES, 3));
decode_fifo({GX_CMD_INVL_VC});
EXPECT_TRUE(g_gxState.stateDirty);
decode_fifo(draw(GX_TRIANGLES, 3));
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
}
TEST_F(GXFifoTest, EqualStrideVertexFormatChangeBreaksDrawMerging) {
aurora::gfx::testing::use_real_vertex_format_helpers(true);
g_gxState.vtxDesc[GX_VA_POS] = GX_DIRECT;
for (const auto format : {GX_VTXFMT0, GX_VTXFMT1}) {
g_gxState.vtxFmts[format].attrs[GX_VA_POS].cnt = GX_POS_XY;
g_gxState.vtxFmts[format].attrs[GX_VA_POS].type = GX_U8;
}
g_gxState.vtxFmts[GX_VTXFMT1].attrs[GX_VA_POS].frac = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
for (const auto format : {GX_VTXFMT0, GX_VTXFMT1}) {
auto bytes = draw(GX_TRIANGLES, 3, format);
bytes.resize(9);
decode_fifo(bytes);
}
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
}
TEST_F(GXFifoTest, SingleExpandedPrimitiveCannotMergeWithTriangles) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_POINTS, 1));
decode_fifo(draw(GX_TRIANGLES, 3));
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
}
TEST(StagingMapping, RetiredCallbacksCannotPublishAnotherBuffersReadiness) {
using namespace aurora::gfx;
StagingMapState state;
const auto old = state.request();
EXPECT_EQ(state.request(), 0u);
state.reset();
const auto current = state.request();
EXPECT_FALSE(state.complete(old, BufferMapState::Mapped));
EXPECT_FALSE(state.complete(old, BufferMapState::Unmapped));
EXPECT_EQ(state.state(), BufferMapState::Mapping);
EXPECT_TRUE(state.complete(current, BufferMapState::Mapped));
EXPECT_FALSE(state.complete(current, BufferMapState::Unmapped));
EXPECT_EQ(state.state(), BufferMapState::Mapped);
}
TEST(StagingMapping, AsyncCompletionWakesWaiters) {
using namespace aurora::gfx;
StagingMapState state;
const auto generation = state.request();
std::thread callback([&] {
std::this_thread::sleep_for(std::chrono::milliseconds(10));
state.complete(generation, BufferMapState::Mapped);
});
const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(2);
while (state.state() == BufferMapState::Mapping && std::chrono::steady_clock::now() < deadline)
state.wait_for_progress();
callback.join();
EXPECT_EQ(state.state(), BufferMapState::Mapped);
}
+4
View File
@@ -1551,6 +1551,10 @@ int RuntimeMain(int argc, char** argv) {
WiiRemoteInput::ConfigureSdlHints(RuntimeConfigFile::WiiRemotesEnabled(true));
const AuroraInfo auroraInfo = aurora_initialize(0, nullptr, &auroraConfig);
if (auroraInfo.initializationStatus != AURORA_INITIALIZATION_SUCCESS) {
throw std::runtime_error(auroraInfo.initializationError != nullptr
? auroraInfo.initializationError : "No supported graphics backend is available");
}
if (requestedBackend != BACKEND_AUTO && auroraInfo.backend != requestedBackend) {
RT_LOG(RT_TAG_RUNTIME) << "graphics_api=\"" << backend
<< "\" is not available on this system; aurora fell back to \""