aurora: add asynchronous OpenXR stereo replay bridge

This commit is contained in:
iChris4 committed 2026-09-03 04:00:02 +02:00
1 parent 8c8e4561e9
commit d8788de919
17 files changed
+1521 -74

No files matched your search

+7
View File
@@ -33,6 +33,13 @@ endif ()
if (AURORA_ENABLE_GX) if (AURORA_ENABLE_GX)
target_compile_definitions(aurora_core PUBLIC AURORA_ENABLE_GX WEBGPU_DAWN) target_compile_definitions(aurora_core PUBLIC AURORA_ENABLE_GX WEBGPU_DAWN)
target_sources(aurora_core PRIVATE lib/webgpu/gpu.cpp lib/webgpu/gpu_cache.cpp lib/dawn/BackendBinding.cpp) target_sources(aurora_core PRIVATE lib/webgpu/gpu.cpp lib/webgpu/gpu_cache.cpp lib/dawn/BackendBinding.cpp)
# The translation unit supplies C ABI fallback stubs when Dawn D3D12 is
# unavailable. Keep those symbols linkable on every Windows GX build so an
# OpenXR-enabled runtime can fail back to desktop mode at runtime instead
# of producing unresolved interop references.
if (CMAKE_SYSTEM_NAME STREQUAL Windows)
target_sources(aurora_core PRIVATE lib/webgpu/d3d12_interop.cpp)
endif ()
target_link_libraries(aurora_core PRIVATE dawn::webgpu_dawn) target_link_libraries(aurora_core PRIVATE dawn::webgpu_dawn)
if (DAWN_ENABLE_VULKAN) if (DAWN_ENABLE_VULKAN)
target_compile_definitions(aurora_core PRIVATE DAWN_ENABLE_BACKEND_VULKAN) target_compile_definitions(aurora_core PRIVATE DAWN_ENABLE_BACKEND_VULKAN)
+32 -4
View File
@@ -82,14 +82,17 @@ enum { AURORA_STEREO_EYE_COUNT = 2 };
/** /**
* One eye of a stereo frame supplied by the host application. * One eye of a stereo frame supplied by the host application.
* *
* projection is a row-major, renderer-ready 4x4 projection matrix. It uses * projection is row-major and supplies the OpenXR frustum's X/Y scale and
* the same convention as the matrix uploaded by GXSetProjection after * asymmetric-center terms at [0][0], [0][2], [1][1], and [1][2]. Aurora
* Aurora's depth-range adjustment. * applies those four values to each perspective GX draw while preserving the
* draw's own depth mapping and renderer depth-range adjustment.
* *
* viewFromCenter is a row-major affine 3x4 transform from the game's * viewFromCenter is a row-major affine 3x4 transform from the game's
* recorded center-eye view space into this eye's view space. Identity keeps * recorded center-eye view space into this eye's view space. Identity keeps
* the recorded view and is useful when the game has already applied the eye * the recorded view and is useful when the game has already applied the eye
* transform before issuing GX commands. * transform before issuing GX commands.
*
* Both transforms are ignored in AURORA_STEREO_FRAME_VIRTUAL_SCREEN mode.
*/ */
typedef struct { typedef struct {
uint32_t width; uint32_t width;
@@ -98,13 +101,30 @@ typedef struct {
float viewFromCenter[12]; float viewFromCenter[12];
} AuroraStereoEye; } AuroraStereoEye;
typedef enum {
// Replay perspective GX draws with the supplied per-eye transforms.
AURORA_STEREO_FRAME_IMMERSIVE_REPLAY = 0,
// Copy the completed mono present source to both eye outputs. The OpenXR
// backend can present these images as a compositor quad layer.
AURORA_STEREO_FRAME_VIRTUAL_SCREEN = 1,
} AuroraStereoFrameMode;
// aurora_end_frame() uses this sentinel when its caller cannot associate a
// sealed frame with an application safety state. Immersive providers are only
// accepted through aurora_end_frame_tagged() with an exact matching tag.
#define AURORA_STEREO_CONTENT_TAG_UNKNOWN UINT64_MAX
/** /**
* Stereo data for one sealed GX frame. frameToken is opaque to Aurora and is * Stereo data for one sealed GX frame. frameToken is opaque to Aurora and is
* forwarded unchanged to the internal stereo output sink. * forwarded unchanged to the internal stereo output sink. contentTag must
* match the tag latched by aurora_end_frame_tagged() for immersive replay.
*/ */
typedef struct { typedef struct {
uint64_t frameToken; uint64_t frameToken;
AuroraStereoEye eyes[AURORA_STEREO_EYE_COUNT]; AuroraStereoEye eyes[AURORA_STEREO_EYE_COUNT];
// Appended to preserve the frameToken/eyes prefix used by older providers.
AuroraStereoFrameMode mode;
uint64_t contentTag;
} AuroraStereoFrame; } AuroraStereoFrame;
/** /**
@@ -186,6 +206,9 @@ void aurora_shutdown();
const AuroraEvent* aurora_update(); const AuroraEvent* aurora_update();
bool aurora_begin_frame(); bool aurora_begin_frame();
void aurora_end_frame(); void aurora_end_frame();
// Seal the current frame with an opaque application safety tag. Aurora rejects
// an immersive provider packet unless its contentTag matches this exact frame.
void aurora_end_frame_tagged(uint64_t contentTag);
typedef void (*AuroraFrameWorkerWaitCallback)(); typedef void (*AuroraFrameWorkerWaitCallback)();
// Called from the producer thread at bounded intervals while Aurora waits for // Called from the producer thread at bounded intervals while Aurora waits for
// the asynchronous frame worker. The callback must not enter Aurora. // the asynchronous frame worker. The callback must not enter Aurora.
@@ -195,6 +218,11 @@ void aurora_set_frame_worker_wait_callback(AuroraFrameWorkerWaitCallback callbac
void aurora_set_stereo_frame_provider(AuroraStereoFrameProvider provider, void* userdata); void aurora_set_stereo_frame_provider(AuroraStereoFrameProvider provider, void* userdata);
void aurora_wait_for_frame_worker(); void aurora_wait_for_frame_worker();
bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros); bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros);
// Producer-thread shutdown barrier. If an asynchronous cycle is waiting for
// the next begin-frame permission, grant that permission and wait until the
// worker is fully done. Unlike aurora_wait_for_frame_worker(), this is safe in
// the gap between aurora_end_frame() and aurora_begin_frame().
void aurora_quiesce_frame_worker();
// Absolute schedule for the next sealed frame, on steady_clock: baseNanos anchors the group and // Absolute schedule for the next sealed frame, on steady_clock: baseNanos anchors the group and
// intervalNanos is the period, so slot k of N+1 fires at base + k*interval/(N+1). Zeros clear it. // intervalNanos is the period, so slot k of N+1 fires at base + k*interval/(N+1). Zeros clear it.
void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos); void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos);
@@ -0,0 +1,88 @@
#ifndef AURORA_D3D12_INTEROP_H
#define AURORA_D3D12_INTEROP_H
#ifdef __cplusplus
#include <cstdint>
extern "C" {
#else
#include "stdbool.h"
#include "stdint.h"
#endif
enum { AURORA_D3D12_STEREO_MAX_TARGETS = 2 };
/**
* Borrowed native objects owned by Aurora/Dawn. They remain valid until
* aurora_shutdown() and must not be released by the caller.
*
* colorDxgiFormat is the DXGI format matching Aurora's single-sample eye
* output. adapterLuid is returned in the same split representation used by
* OpenXR's XrGraphicsRequirementsD3D12KHR.
*/
typedef struct {
void* device;
void* queue;
int64_t colorDxgiFormat;
uint32_t adapterLuidLow;
int32_t adapterLuidHigh;
} AuroraD3D12NativeHandles;
/** One acquired OpenXR swapchain image for the next Aurora stereo sink. */
typedef struct {
void* resource;
uint32_t width;
uint32_t height;
int64_t dxgiFormat;
} AuroraD3D12StereoTarget;
/**
* Fired when Aurora either finishes or abandons the stereo sink. `success`
* guarantees that the final same-queue D3D12 copy and its completion fence were
* enqueued. A false result may occur after ExecuteCommandLists, so callers must
* conservatively retain externally owned targets until graphics/session
* teardown. The callback must not wait for the GPU or re-enter Aurora.
*/
typedef void (*AuroraD3D12StereoSubmittedCallback)(uint64_t frameToken, bool success,
void* userdata);
/** Returns false unless the active Aurora backend is Dawn D3D12. */
bool aurora_d3d12_get_native_handles(AuroraD3D12NativeHandles* handles);
/**
* Installs the internal zero-readback stereo sink. Call while Aurora's frame
* worker is idle, after aurora_initialize().
*/
bool aurora_d3d12_enable_stereo_bridge(AuroraD3D12StereoSubmittedCallback submitted,
void* userdata);
/**
* Publishes the acquired OpenXR image(s) for frameToken. Immersive projection
* frames supply two targets; virtual-screen quad frames supply one. Exactly
* one frame may be pending at a time.
*/
bool aurora_d3d12_set_stereo_targets(uint64_t frameToken,
const AuroraD3D12StereoTarget* targets,
uint32_t targetCount);
/**
* Withdraws frameToken only while its target has not been encoded. This is
* safe to race with Aurora's frame worker: false means the worker already owns
* encoded work (or the token is no longer pending), so the submitted callback
* remains the only completion authority. A successful cancellation performs
* no GPU work and deliberately does not fire the callback.
*/
bool aurora_d3d12_cancel_stereo_targets(uint64_t frameToken);
/**
* Removes the sink and drains bridge resources. The worker must be idle. Returns
* false when queued work cannot be fenced; in that case the bridge is retained
* for the process lifetime and the caller must likewise retain its graphics/XR
* owners rather than destroying resources with unknown GPU use.
*/
bool aurora_d3d12_disable_stereo_bridge();
#ifdef __cplusplus
}
#endif
#endif
+13
View File
@@ -82,6 +82,19 @@ uint32_t aurora_get_queued_pipeline_count();
void aurora_set_disable_copy_filter(bool disabled); void aurora_set_disable_copy_filter(bool disabled);
bool aurora_get_disable_copy_filter(); bool aurora_get_disable_copy_filter();
// Immersive (stereo) replay EFB controls, both enabled by default, and both
// live: they take effect on the next frame with no restart.
//
// stop_at_display_copy ends each eye's replay at the frame's final GXCopyDisp,
// so an eye holds exactly the image the game presented. skip_copy_clears drops
// the EFB reset a GX copy performs after copying, which on the Wii prepares the
// reused EFB for the next frame but on a per-frame eye attachment only erases
// the replay. Disable either to compare against the raw replay.
void aurora_set_stereo_stop_at_display_copy(bool enabled);
bool aurora_get_stereo_stop_at_display_copy();
void aurora_set_stereo_skip_copy_clears(bool enabled);
bool aurora_get_stereo_skip_copy_clears();
// Guest-RAM write tracking. `generation` changes whenever guest RAM covering a host range was // Guest-RAM write tracking. `generation` changes whenever guest RAM covering a host range was
// written (or returns AURORA_GUEST_WRITE_UNTRACKED); `notify` reports writes aurora made itself. // written (or returns AURORA_GUEST_WRITE_UNTRACKED); `notify` reports writes aurora made itself.
#define AURORA_GUEST_WRITE_UNTRACKED UINT64_MAX #define AURORA_GUEST_WRITE_UNTRACKED UINT64_MAX
+257 -40
View File
@@ -35,6 +35,7 @@
#include <condition_variable> #include <condition_variable>
#include <cstdio> #include <cstdio>
#include <cstdlib> #include <cstdlib>
#include <cstring>
#include <deque> #include <deque>
#include <fstream> #include <fstream>
#include <memory> #include <memory>
@@ -75,6 +76,7 @@ struct StereoProviderRegistration {
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
struct StereoSinkRegistration { struct StereoSinkRegistration {
stereo::SinkCallback callback = nullptr; stereo::SinkCallback callback = nullptr;
stereo::SubmitCallback submitted = nullptr;
void* userdata = nullptr; void* userdata = nullptr;
}; };
#endif #endif
@@ -242,7 +244,7 @@ enum class ImGuiFramePolicy {
bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy = ImGuiFramePolicy::Immediate, bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy = ImGuiFramePolicy::Immediate,
bool* imguiNewFrameOwed = nullptr) noexcept; bool* imguiNewFrameOwed = nullptr) noexcept;
bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewFrameOwed) noexcept; bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewFrameOwed) noexcept;
void end_frame_impl(bool pumpEvents, bool drainFifo) noexcept; void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag) noexcept;
// The two publication points of a frame-worker cycle, cleared together under `mutex`. Sealed: // The two publication points of a frame-worker cycle, cleared together under `mutex`. Sealed:
// producer-shared renderer state is free again. Done: slots encoded, presented, ImGui restarted. // producer-shared renderer state is free again. Done: slots encoded, presented, ImGui restarted.
@@ -259,6 +261,9 @@ struct FrameWorkerState {
bool started = false; bool started = false;
bool stop = false; bool stop = false;
bool jobPending = false; bool jobPending = false;
// Written with jobPending and copied by the worker under this mutex. It
// belongs to that exact queued frame, not to the producer's next frame.
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
// Readiness is polled thousands of times per frame, so these flags double as a publication // Readiness is polled thousands of times per frame, so these flags double as a publication
// barrier. `sealed` is released before `ready`, and both are cleared under `mutex`. // barrier. `sealed` is released before `ready`, and both are cleared under `mutex`.
std::atomic_bool sealed{true}; std::atomic_bool sealed{true};
@@ -299,7 +304,7 @@ bool frame_worker_requested() noexcept {
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
// Returns false when a stop request was observed mid-cycle. // Returns false when a stop request was observed mid-cycle.
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame) noexcept; bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag) noexcept;
#endif #endif
void frame_worker_main() noexcept { void frame_worker_main() noexcept {
@@ -315,21 +320,26 @@ void frame_worker_main() noexcept {
#endif #endif
for (;;) { for (;;) {
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
{ {
std::unique_lock lock(g_frameWorker.mutex); std::unique_lock lock(g_frameWorker.mutex);
g_frameWorker.cv.wait(lock, [] { return g_frameWorker.stop || g_frameWorker.jobPending; }); g_frameWorker.cv.wait(lock, [] { return g_frameWorker.stop || g_frameWorker.jobPending; });
if (g_frameWorker.stop) { if (g_frameWorker.stop) {
break; break;
} }
contentTag = g_frameWorker.contentTag;
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
g_frameWorker.jobPending = false; g_frameWorker.jobPending = false;
} }
// The CPU already decoded the sealed frame at its GX boundary; the worker only owns // The CPU already decoded the sealed frame at its GX boundary; the worker only owns
// encode/submit/present, so it never touches the producer's next FIFO buffer. // encode/submit/present, so it never touches the producer's next FIFO buffer.
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
if (!run_frame_worker_cycle(sealedFrame)) { if (!run_frame_worker_cycle(sealedFrame, contentTag)) {
break; break;
} }
#else
(void)contentTag;
#endif #endif
} }
@@ -353,6 +363,7 @@ void ensure_frame_worker_started() noexcept {
} }
g_frameWorker.stop = false; g_frameWorker.stop = false;
g_frameWorker.jobPending = false; g_frameWorker.jobPending = false;
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
g_frameWorker.sealed.store(true, std::memory_order_release); g_frameWorker.sealed.store(true, std::memory_order_release);
g_frameWorker.ready.store(true, std::memory_order_release); g_frameWorker.ready.store(true, std::memory_order_release);
g_frameWorker.prepareAllowed = false; g_frameWorker.prepareAllowed = false;
@@ -421,6 +432,8 @@ void stop_frame_worker() noexcept {
g_frameWorker.started = false; g_frameWorker.started = false;
g_frameWorker.threadId = {}; g_frameWorker.threadId = {};
g_frameWorker.framePrepared = false; g_frameWorker.framePrepared = false;
g_frameWorker.jobPending = false;
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
} }
uint32_t align_to(uint32_t value, uint32_t alignment) noexcept { uint32_t align_to(uint32_t value, uint32_t alignment) noexcept {
@@ -496,6 +509,11 @@ struct StereoEyeTarget {
webgpu::TextureWithSampler color; webgpu::TextureWithSampler color;
webgpu::TextureWithSampler resolvedColor; webgpu::TextureWithSampler resolvedColor;
webgpu::TextureWithSampler depth; webgpu::TextureWithSampler depth;
uint32_t requestedWidth = 0;
uint32_t requestedHeight = 0;
uint32_t samples = 0;
wgpu::TextureFormat colorFormat = wgpu::TextureFormat::Undefined;
wgpu::TextureFormat depthFormat = wgpu::TextureFormat::Undefined;
const webgpu::TextureWithSampler& output() const noexcept { const webgpu::TextureWithSampler& output() const noexcept {
return resolvedColor.texture ? resolvedColor : color; return resolvedColor.texture ? resolvedColor : color;
@@ -506,8 +524,9 @@ std::array<StereoEyeTarget, AURORA_STEREO_EYE_COUNT> g_stereoEyeTargets;
void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height) { void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height) {
auto& target = g_stereoEyeTargets[eyeIndex]; auto& target = g_stereoEyeTargets[eyeIndex];
const uint32_t samples = webgpu::g_graphicsConfig.msaaSamples; const uint32_t samples = webgpu::g_graphicsConfig.msaaSamples;
if (target.color.texture && target.color.size.width == width && target.color.size.height == height && if (target.color.texture && target.requestedWidth == width && target.requestedHeight == height &&
((samples > 1 && target.resolvedColor.texture) || (samples == 1 && !target.resolvedColor.texture))) { target.samples == samples && target.colorFormat == webgpu::g_graphicsConfig.surfaceConfiguration.format &&
target.depthFormat == webgpu::g_graphicsConfig.depthFormat) {
return; return;
} }
@@ -531,9 +550,15 @@ void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height
target.depth.view = target.depth.texture.CreateView(); target.depth.view = target.depth.texture.CreateView();
target.depth.size = target.color.size; target.depth.size = target.color.size;
target.depth.format = webgpu::g_graphicsConfig.depthFormat; target.depth.format = webgpu::g_graphicsConfig.depthFormat;
target.requestedWidth = width;
target.requestedHeight = height;
target.samples = samples;
target.colorFormat = target.color.format;
target.depthFormat = target.depth.format;
} }
std::optional<AuroraStereoFrame> request_stereo_frame(uint32_t logicalFrame) noexcept { std::optional<AuroraStereoFrame> request_stereo_frame(uint32_t logicalFrame,
uint64_t contentTag) noexcept {
StereoProviderRegistration registration; StereoProviderRegistration registration;
{ {
std::lock_guard lock(g_stereoRegistrationMutex); std::lock_guard lock(g_stereoRegistrationMutex);
@@ -554,14 +579,36 @@ std::optional<AuroraStereoFrame> request_stereo_frame(uint32_t logicalFrame) noe
if (!provided) { if (!provided) {
return std::nullopt; return std::nullopt;
} }
if (frame.mode != AURORA_STEREO_FRAME_IMMERSIVE_REPLAY && frame.mode != AURORA_STEREO_FRAME_VIRTUAL_SCREEN) {
Log.warn("Stereo frame {} has invalid mode {}; rendering in mono", logicalFrame,
static_cast<uint32_t>(frame.mode));
return std::nullopt;
}
// A virtual-screen packet only copies the completed mono image and remains
// a safe fallback across a transition. Immersive replay changes the GX
// transforms, so it requires the exact tag latched for this sealed content.
if (frame.mode == AURORA_STEREO_FRAME_IMMERSIVE_REPLAY &&
(contentTag == AURORA_STEREO_CONTENT_TAG_UNKNOWN || frame.contentTag != contentTag)) {
Log.debug("Stereo frame {} content tag does not match its sealed frame; rendering in mono",
logicalFrame);
return std::nullopt;
}
const auto finite = [](const float* values, size_t count) { const auto finite = [](const float* values, size_t count) {
return std::all_of(values, values + count, [](float value) { return std::isfinite(value); }); // aurora_core is built with -ffast-math, so std::isfinite may be folded to
// true. Inspecting the IEEE-754 exponent keeps the provider boundary safe
// under the target's real release flags.
return std::all_of(values, values + count, [](const float& value) {
uint32_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
return (bits & 0x7f800000u) != 0x7f800000u;
});
}; };
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) { for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
const auto& input = frame.eyes[eye]; const auto& input = frame.eyes[eye];
if (input.width == 0 || input.height == 0 || !finite(input.projection, 16) || const bool transformsValid = frame.mode == AURORA_STEREO_FRAME_VIRTUAL_SCREEN ||
!finite(input.viewFromCenter, 12)) { (finite(input.projection, 16) && finite(input.viewFromCenter, 12));
if (input.width == 0 || input.height == 0 || !transformsValid) {
Log.warn("Stereo frame {} has invalid eye {} dimensions or transforms; rendering in mono", Log.warn("Stereo frame {} has invalid eye {} dimensions or transforms; rendering in mono",
logicalFrame, eye); logicalFrame, eye);
return std::nullopt; return std::nullopt;
@@ -593,19 +640,58 @@ gfx::StereoReplayFrame make_stereo_replay_frame(const AuroraStereoFrame& input)
return replay; return replay;
} }
void run_stereo_sink(wgpu::CommandEncoder& encoder, uint64_t frameToken, uint32_t logicalFrame) noexcept { void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::PresentSource& source,
uint32_t eyeIndex) {
const auto& output = g_stereoEyeTargets[eyeIndex].output();
const std::array attachments{
wgpu::RenderPassColorAttachment{
.view = output.view,
.loadOp = wgpu::LoadOp::Clear,
.storeOp = wgpu::StoreOp::Store,
.clearValue = {.r = 0.0, .g = 0.0, .b = 0.0, .a = 1.0},
},
};
const wgpu::RenderPassDescriptor descriptor{
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
};
{
const auto pass = encoder.BeginRenderPass(&descriptor);
if (source.bindGroup && source.size.width != 0 && source.size.height != 0) {
const auto viewport = webgpu::calculate_present_viewport(
output.size.width, output.size.height, source.size.width, source.size.height);
pass.SetPipeline(webgpu::g_CopyPipeline);
pass.SetBindGroup(0, source.bindGroup, 0, nullptr);
pass.SetViewport(viewport.left, viewport.top, viewport.width, viewport.height,
viewport.znear, viewport.zfar);
pass.Draw(3);
}
pass.End();
}
}
struct PendingStereoSink {
stereo::SinkFrame frame;
stereo::SubmitCallback submitted = nullptr;
void* userdata = nullptr;
};
std::optional<PendingStereoSink> run_stereo_sink(wgpu::CommandEncoder& encoder, uint64_t frameToken,
uint32_t logicalFrame, AuroraStereoFrameMode mode) noexcept {
StereoSinkRegistration registration; StereoSinkRegistration registration;
{ {
std::lock_guard lock(g_stereoRegistrationMutex); std::lock_guard lock(g_stereoRegistrationMutex);
registration = g_stereoSink; registration = g_stereoSink;
} }
if (registration.callback == nullptr) { if (registration.callback == nullptr) {
return; return std::nullopt;
} }
stereo::SinkFrame frame{ stereo::SinkFrame frame{
.frameToken = frameToken, .frameToken = frameToken,
.logicalFrame = logicalFrame, .logicalFrame = logicalFrame,
.mode = mode,
}; };
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) { for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
const auto& output = g_stereoEyeTargets[eye].output(); const auto& output = g_stereoEyeTargets[eye].output();
@@ -616,7 +702,14 @@ void run_stereo_sink(wgpu::CommandEncoder& encoder, uint64_t frameToken, uint32_
.format = output.format, .format = output.format,
}; };
} }
registration.callback(encoder, frame, registration.userdata); if (!registration.callback(encoder, frame, registration.userdata)) {
return std::nullopt;
}
return PendingStereoSink{
.frame = frame,
.submitted = registration.submitted,
.userdata = registration.userdata,
};
} }
void request_surface_reconfigure() noexcept { void request_surface_reconfigure() noexcept {
@@ -1368,11 +1461,19 @@ void shutdown() noexcept {
stop_frame_worker(); stop_frame_worker();
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
stop_presenter(); stop_presenter();
g_stereoEyeTargets = {};
g_presentationImagePools = {}; g_presentationImagePools = {};
imgui::shutdown(); imgui::shutdown();
gfx::shutdown(); gfx::shutdown();
webgpu::shutdown(); webgpu::shutdown();
#endif #endif
{
std::lock_guard lock(g_stereoRegistrationMutex);
g_stereoProvider = {};
#ifdef AURORA_ENABLE_GX
g_stereoSink = {};
#endif
}
input::shutdown(); input::shutdown();
window::shutdown(); window::shutdown();
} }
@@ -1473,6 +1574,10 @@ bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewF
struct SealedFrameContext { struct SealedFrameContext {
wgpu::CommandEncoder encoder; // slot 0's encoder; already holds the staging copies wgpu::CommandEncoder encoder; // slot 0's encoder; already holds the staging copies
webgpu::PresentSource presentSource{}; webgpu::PresentSource presentSource{};
std::optional<gfx::StereoReplayFrame> stereoReplay;
uint64_t stereoFrameToken = 0;
AuroraStereoFrameMode stereoFrameMode = AURORA_STEREO_FRAME_IMMERSIVE_REPLAY;
bool immersiveStereoPrepared = false;
uint64_t scheduleBaseNanos = 0; uint64_t scheduleBaseNanos = 0;
uint64_t scheduleIntervalNanos = 0; uint64_t scheduleIntervalNanos = 0;
uint32_t interpolatedFrameCount = 0; uint32_t interpolatedFrameCount = 0;
@@ -1485,7 +1590,8 @@ struct SealedFrameContext {
// Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and // Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and
// a FIFO already drained into the recorded pass list. // a FIFO already drained into the recorded pass list.
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx) { void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx,
uint64_t contentTag) {
ZoneScopedN("Seal frame"); ZoneScopedN("Seal frame");
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{ const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
.label = "Redraw encoder", .label = "Redraw encoder",
@@ -1494,7 +1600,32 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx) {
// Probe-sized CPU-consumed copies read back asynchronously. Their downscale blits push uniforms, // Probe-sized CPU-consumed copies read back asynchronously. Their downscale blits push uniforms,
// so prepare them while the producer's staging buffers are still mapped. // so prepare them while the producer's staging buffers are still mapped.
gfx::efb_ram::seal_async_downloads(); gfx::efb_ram::seal_async_downloads();
gfx::end_frame(ctx.encoder); // current_frame() advances inside gfx::end_frame; unsigned wrap maps the
// pre-first-frame UINT32_MAX value to logical frame zero.
ctx.logicalFrame = gfx::current_frame() + 1;
if (const auto stereoInput = request_stereo_frame(ctx.logicalFrame, contentTag)) {
ctx.stereoFrameToken = stereoInput->frameToken;
ctx.stereoFrameMode = stereoInput->mode;
ctx.stereoReplay = make_stereo_replay_frame(*stereoInput);
}
if (ctx.stereoReplay && ctx.stereoFrameMode == AURORA_STEREO_FRAME_IMMERSIVE_REPLAY) {
// Keep the accepted frame alive even if the additional eye-uniform copies
// do not fit. The encode phase will duplicate the completed mono image to
// both eyes, allowing the sink to release/end the already-acquired XR
// frame instead of orphaning it indefinitely.
ctx.immersiveStereoPrepared = gfx::end_frame(ctx.encoder, *ctx.stereoReplay);
static bool immersiveReplaySuccessLogged = false;
static bool immersiveReplayFallbackLogged = false;
if (ctx.immersiveStereoPrepared && !immersiveReplaySuccessLogged) {
immersiveReplaySuccessLogged = true;
Log.info("Immersive stereo GX replay prepared successfully for both eyes");
} else if (!ctx.immersiveStereoPrepared && !immersiveReplayFallbackLogged) {
immersiveReplayFallbackLogged = true;
Log.warn("Immersive stereo GX replay preparation failed; duplicating the mono image into the projection layer");
}
} else {
gfx::end_frame(ctx.encoder);
}
gfx::g_stats.presentedFrameCount = 0; gfx::g_stats.presentedFrameCount = 0;
gfx::g_stats.interpolatedFrameCount = 0; gfx::g_stats.interpolatedFrameCount = 0;
// Latched before the producer's next gfx::begin_frame() calls // Latched before the producer's next gfx::begin_frame() calls
@@ -1540,33 +1671,40 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
return {}; return {};
} }
if (!ctx.interpolationActive) { if (!ctx.interpolationActive) {
return PresentClock::time_point{std::chrono::nanoseconds{ return PresentClock::time_point{
ctx.scheduleBaseNanos - ctx.scheduleIntervalNanos + kNativePresentOffsetNanos}}; std::chrono::nanoseconds{ctx.scheduleBaseNanos - ctx.scheduleIntervalNanos + kNativePresentOffsetNanos}};
} }
const uint64_t offsetNanos = const uint64_t offsetNanos = (ctx.scheduleIntervalNanos * static_cast<uint64_t>(slot)) / presentationJobCount;
(ctx.scheduleIntervalNanos * static_cast<uint64_t>(slot)) / presentationJobCount;
return PresentClock::time_point{std::chrono::nanoseconds{ctx.scheduleBaseNanos + offsetNanos}}; return PresentClock::time_point{std::chrono::nanoseconds{ctx.scheduleBaseNanos + offsetNanos}};
}; };
std::vector<PresentationJob> presentationJobs; std::vector<PresentationJob> presentationJobs;
presentationJobs.reserve(presentationJobCount); presentationJobs.reserve(presentationJobCount);
std::optional<PendingStereoSink> pendingStereoSink;
// Each slot is submitted as soon as it is encoded, so the GPU starts slot 0 while slot 1 is still // Each slot is submitted as soon as it is encoded, so the GPU starts slot 0 while slot 1 is still
// recording. Queue order preserves the ordering the single batched buffer gave. // recording. Queue order preserves the ordering the single batched buffer gave.
const wgpu::CommandBufferDescriptor cmdBufDescriptor{ const wgpu::CommandBufferDescriptor cmdBufDescriptor{
.label = "Presentation slot command buffer", .label = "Presentation slot command buffer",
}; };
const auto submitEncodedSlot = [&](wgpu::CommandEncoder& target) { const auto submitEncodedSlot = [&](wgpu::CommandEncoder& target,
const PendingStereoSink* stereoSubmission = nullptr) {
const auto buffer = target.Finish(&cmdBufDescriptor); const auto buffer = target.Finish(&cmdBufDescriptor);
std::lock_guard submitLock(g_queueSubmitMutex); {
g_queue.Submit(1, &buffer); std::lock_guard submitLock(g_queueSubmitMutex);
g_queue.Submit(1, &buffer);
// The native backend may enqueue follow-up work on Dawn's underlying
// graphics queue. Keep it adjacent to this submission so presenter or
// EFB work cannot interleave between the WebGPU copy and that handoff.
if (stereoSubmission != nullptr && stereoSubmission->submitted != nullptr) {
stereoSubmission->submitted(stereoSubmission->frame, stereoSubmission->userdata);
}
}
}; };
if (ctx.replayInterpolatedFrames) { if (ctx.replayInterpolatedFrames) {
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
++interpolatedFrame) {
gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false); gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false);
auto image = auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true); encode_presentation_snapshot(encoder, ctx.presentSource, *image, true);
presentationJobs.push_back({ presentationJobs.push_back({
.image = std::move(image), .image = std::move(image),
@@ -1581,15 +1719,15 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed // A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots. // stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
gfx::render(sealedFrame, encoder, -1, true); const bool stereoOutput = ctx.stereoReplay.has_value();
const bool immersiveReplay = stereoOutput && ctx.immersiveStereoPrepared;
gfx::render(sealedFrame, encoder, -1, !immersiveReplay);
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder; // The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
// completion is harvested in gfx::after_submit, never waited on here. // completion is harvested in gfx::after_submit, never waited on here.
gfx::efb_ram::encode_async_downloads(encoder); gfx::efb_ram::encode_async_downloads(encoder);
if (!ctx.replayInterpolatedFrames) { if (!ctx.replayInterpolatedFrames) {
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
++interpolatedFrame) { auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
auto image =
acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true); encode_presentation_snapshot(encoder, ctx.presentSource, *image, true);
presentationJobs.push_back({ presentationJobs.push_back({
.image = std::move(image), .image = std::move(image),
@@ -1602,9 +1740,35 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
encoder = g_device.CreateCommandEncoder(&encoderDescriptor); encoder = g_device.CreateCommandEncoder(&encoderDescriptor);
} }
} }
auto finalImage = auto finalImage = acquire_presentation_image(ctx.interpolatedFrameCount, ctx.snapshotWidth, ctx.snapshotHeight);
acquire_presentation_image(ctx.interpolatedFrameCount, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true); encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true);
// Keep both eye replays and the sink copy in the final submission. The
// duplicate-slot path above may submit and rotate the encoder several
// times, so encoding stereo before it would pair the post-submit callback
// with the wrong command buffer.
if (immersiveReplay) {
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
gfx::render_stereo_eye(sealedFrame, encoder, *ctx.stereoReplay, eye, eye + 1 == AURORA_STEREO_EYE_COUNT);
}
} else if (stereoOutput) {
// Use the completed mono snapshot so virtual-screen XR includes ImGui at
// the same scale and aspect as the desktop presentation. Rendering the
// same ImGui draw data directly into differently-sized eye textures would
// make the backend restore the desktop-sized viewport.
const webgpu::PresentSource completedMono{
.bindGroup = finalImage->bindGroup,
.texture = finalImage->texture.texture,
.size = finalImage->texture.size,
.format = finalImage->texture.format,
};
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
encode_virtual_screen_eye(encoder, completedMono, eye);
}
}
if (stereoOutput) {
pendingStereoSink = run_stereo_sink(encoder, ctx.stereoFrameToken, ctx.logicalFrame, ctx.stereoFrameMode);
}
auto pendingFrameCapture = encode_frame_capture(encoder, ctx.presentSource); auto pendingFrameCapture = encode_frame_capture(encoder, ctx.presentSource);
presentationJobs.push_back({ presentationJobs.push_back({
.image = std::move(finalImage), .image = std::move(finalImage),
@@ -1612,7 +1776,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount), .presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
.interpolated = false, .interpolated = false,
}); });
submitEncodedSlot(encoder); submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
// A group that finished encoding past its anchor slides forward by whole display periods, never // A group that finished encoding past its anchor slides forward by whole display periods, never
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period. // per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
@@ -1719,7 +1883,7 @@ void record_frame_telemetry() {
// One complete frame-worker cycle. The scene encode only leaves the renderer mutex when // One complete frame-worker cycle. The scene encode only leaves the renderer mutex when
// interpolation actually inserts slots; otherwise both phases publish together. // interpolation actually inserts slots; otherwise both phases publish together.
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame) noexcept { bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag) noexcept {
ZoneScopedN("Frame worker cycle"); ZoneScopedN("Frame worker cycle");
webgpu::fail_if_device_lost(); webgpu::fail_if_device_lost();
SealedFrameContext ctx; SealedFrameContext ctx;
@@ -1727,7 +1891,7 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame) noexcept {
bool overlapEncode = false; bool overlapEncode = false;
{ {
std::lock_guard gpuLock(g_rendererGpuMutex); std::lock_guard gpuLock(g_rendererGpuMutex);
seal_frame_locked(sealedFrame, ctx); seal_frame_locked(sealedFrame, ctx, contentTag);
overlapEncode = ctx.interpolationActive; overlapEncode = ctx.interpolationActive;
if (!overlapEncode) { if (!overlapEncode) {
presentationJobs = encode_sealed_frame(sealedFrame, ctx); presentationJobs = encode_sealed_frame(sealedFrame, ctx);
@@ -1787,7 +1951,7 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame) noexcept {
// Synchronous frame submission: seal, encode and present inline on the calling thread. Used when // Synchronous frame submission: seal, encode and present inline on the calling thread. Used when
// the frame worker is disabled (RenderDoc captures) and on the boot path. // the frame worker is disabled (RenderDoc captures) and on the boot path.
void end_frame_impl(bool pumpEvents, bool drainFifo) noexcept { void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag) noexcept {
ZoneScoped; ZoneScoped;
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost(); webgpu::fail_if_device_lost();
@@ -1802,7 +1966,7 @@ void end_frame_impl(bool pumpEvents, bool drainFifo) noexcept {
if (drainFifo) { if (drainFifo) {
gx::fifo::drain(); gx::fifo::drain();
} }
seal_frame_locked(sealedFrame, ctx); seal_frame_locked(sealedFrame, ctx, contentTag);
presentationJobs = encode_sealed_frame(sealedFrame, ctx); presentationJobs = encode_sealed_frame(sealedFrame, ctx);
} }
publish_presentations(std::move(presentationJobs), ctx.interpolationActive); publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
@@ -1810,6 +1974,7 @@ void end_frame_impl(bool pumpEvents, bool drainFifo) noexcept {
#else #else
(void)pumpEvents; (void)pumpEvents;
(void)drainFifo; (void)drainFifo;
(void)contentTag;
#endif #endif
} }
@@ -1878,12 +2043,12 @@ bool begin_frame() noexcept {
return prepared; return prepared;
} }
void end_frame() noexcept { void end_frame(uint64_t contentTag) noexcept {
#ifdef AURORA_ENABLE_GX #ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost(); webgpu::fail_if_device_lost();
#endif #endif
if (!frame_worker_requested()) { if (!frame_worker_requested()) {
end_frame_impl(true, true); end_frame_impl(true, true, contentTag);
return; return;
} }
@@ -1903,6 +2068,7 @@ void end_frame() noexcept {
g_frameWorker.framePrepared = false; g_frameWorker.framePrepared = false;
g_frameWorker.sealed.store(false, std::memory_order_release); g_frameWorker.sealed.store(false, std::memory_order_release);
g_frameWorker.ready.store(false, std::memory_order_release); g_frameWorker.ready.store(false, std::memory_order_release);
g_frameWorker.contentTag = contentTag;
g_frameWorker.jobPending = true; g_frameWorker.jobPending = true;
g_frameWorker.prepareAllowed = false; g_frameWorker.prepareAllowed = false;
} }
@@ -1922,7 +2088,47 @@ std::chrono::nanoseconds wait_for_frame_worker_sealed() noexcept {
bool wait_for_frame_worker_for(std::chrono::microseconds timeout) noexcept { bool wait_for_frame_worker_for(std::chrono::microseconds timeout) noexcept {
return wait_for_frame_worker_private_for(FrameWorkerPhase::Done, timeout); return wait_for_frame_worker_private_for(FrameWorkerPhase::Done, timeout);
} }
void quiesce_frame_worker() noexcept {
{
std::lock_guard lock(g_frameWorker.mutex);
if (!g_frameWorker.started || g_frameWorker.threadId == std::this_thread::get_id() ||
g_frameWorker.ready.load(std::memory_order_acquire)) {
return;
}
// This may race the worker just after it passed the prepare wait. In that
// case the permit is harmless and is cleared after DONE before another job
// can be queued by this producer thread.
g_frameWorker.prepareAllowed = true;
}
g_frameWorker.cv.notify_all();
wait_for_frame_worker_private(FrameWorkerPhase::Done);
{
std::lock_guard lock(g_frameWorker.mutex);
g_frameWorker.prepareAllowed = false;
}
}
std::recursive_mutex& renderer_gpu_mutex() noexcept { return g_rendererGpuMutex; } std::recursive_mutex& renderer_gpu_mutex() noexcept { return g_rendererGpuMutex; }
void set_stereo_frame_provider(AuroraStereoFrameProvider provider, void* userdata) noexcept {
std::lock_guard lock(g_stereoRegistrationMutex);
g_stereoProvider = {
.callback = provider,
.userdata = userdata,
};
}
#ifdef AURORA_ENABLE_GX
namespace stereo {
void set_sink(SinkCallback callback, SubmitCallback submitted, void* userdata) noexcept {
std::lock_guard lock(g_stereoRegistrationMutex);
g_stereoSink = {
.callback = callback,
.submitted = submitted,
.userdata = userdata,
};
}
} // namespace stereo
#endif
} // namespace aurora } // namespace aurora
// C API bindings // C API bindings
@@ -1932,14 +2138,19 @@ AuroraInfo aurora_initialize(int argc, char* argv[], const AuroraConfig* config)
void aurora_shutdown() { aurora::shutdown(); } void aurora_shutdown() { aurora::shutdown(); }
const AuroraEvent* aurora_update() { return aurora::update(); } const AuroraEvent* aurora_update() { return aurora::update(); }
bool aurora_begin_frame() { return aurora::begin_frame(); } bool aurora_begin_frame() { return aurora::begin_frame(); }
void aurora_end_frame() { aurora::end_frame(); } void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN); }
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag); }
void aurora_set_frame_worker_wait_callback(AuroraFrameWorkerWaitCallback callback) { void aurora_set_frame_worker_wait_callback(AuroraFrameWorkerWaitCallback callback) {
aurora::g_frameWorkerWaitCallback.store(callback, std::memory_order_release); aurora::g_frameWorkerWaitCallback.store(callback, std::memory_order_release);
} }
void aurora_set_stereo_frame_provider(AuroraStereoFrameProvider provider, void* userdata) {
aurora::set_stereo_frame_provider(provider, userdata);
}
void aurora_wait_for_frame_worker() { aurora::wait_for_frame_worker(); } void aurora_wait_for_frame_worker() { aurora::wait_for_frame_worker(); }
bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros) { bool aurora_wait_for_frame_worker_for(uint32_t timeoutMicros) {
return aurora::wait_for_frame_worker_for(std::chrono::microseconds(timeoutMicros)); return aurora::wait_for_frame_worker_for(std::chrono::microseconds(timeoutMicros));
} }
void aurora_quiesce_frame_worker() { aurora::quiesce_frame_worker(); }
void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos) { void aurora_set_present_schedule(uint64_t baseNanos, uint64_t intervalNanos) {
aurora::g_presentScheduleBaseNanos.store(baseNanos, std::memory_order_release); aurora::g_presentScheduleBaseNanos.store(baseNanos, std::memory_order_release);
aurora::g_presentScheduleIntervalNanos.store(intervalNanos, std::memory_order_release); aurora::g_presentScheduleIntervalNanos.store(intervalNanos, std::memory_order_release);
@@ -2068,6 +2279,12 @@ void aurora_set_log_level(AuroraLogLevel level) { aurora::g_config.logLevel = le
void aurora_set_pause_on_focus_lost(bool value) { aurora::g_config.pauseOnFocusLost = value; } void aurora_set_pause_on_focus_lost(bool value) { aurora::g_config.pauseOnFocusLost = value; }
void aurora_set_disable_copy_filter(bool disabled) { aurora::g_config.disableCopyFilter = disabled; } void aurora_set_disable_copy_filter(bool disabled) { aurora::g_config.disableCopyFilter = disabled; }
bool aurora_get_disable_copy_filter() { return aurora::g_config.disableCopyFilter; } bool aurora_get_disable_copy_filter() { return aurora::g_config.disableCopyFilter; }
void aurora_set_stereo_stop_at_display_copy(bool enabled) {
aurora::gfx::set_stereo_stop_at_display_copy(enabled);
}
bool aurora_get_stereo_stop_at_display_copy() { return aurora::gfx::get_stereo_stop_at_display_copy(); }
void aurora_set_stereo_skip_copy_clears(bool enabled) { aurora::gfx::set_stereo_skip_copy_clears(enabled); }
bool aurora_get_stereo_skip_copy_clears() { return aurora::gfx::get_stereo_skip_copy_clears(); }
void aurora_set_background_input(bool value) { void aurora_set_background_input(bool value) {
aurora::g_config.allowJoystickBackgroundEvents = value; aurora::g_config.allowJoystickBackgroundEvents = value;
aurora::window::set_background_input(value); aurora::window::set_background_input(value);
@@ -446,6 +446,7 @@ void GXCopyDisp(void* dest, GXBool clear) {
static_cast<float>(rect.height) / std::max<float>(g_gxState.dispCopySrc.height, 1.0f), static_cast<float>(rect.height) / std::max<float>(g_gxState.dispCopySrc.height, 1.0f),
(g_gxState.copyClamp & GX_CLAMP_TOP) != 0, (g_gxState.copyClamp & GX_CLAMP_TOP) != 0,
(g_gxState.copyClamp & GX_CLAMP_BOTTOM) != 0); (g_gxState.copyClamp & GX_CLAMP_BOTTOM) != 0);
aurora::gfx::mark_last_resolve_as_display_copy();
aurora::gx::set_display_copy_present_source(); aurora::gx::set_display_copy_present_source();
} }
+4
View File
@@ -11,6 +11,10 @@ struct DrawData {
wgpu::Color color; wgpu::Color color;
float depth = 0.f; float depth = 0.f;
bool useScissor = false; bool useScissor = false;
// Set for the EFB clear a GX copy performs after copying. Immersive replay
// can drop these: an eye attachment is not the Wii's reused EFB, so the
// post-copy reset would erase the image the copy just published.
bool copyClear = false;
ClipRect scissor{}; ClipRect scissor{};
}; };
+261 -27
View File
@@ -20,6 +20,7 @@
#include <chrono> #include <chrono>
#include <cmath> #include <cmath>
#include <cstdlib> #include <cstdlib>
#include <iterator>
#include <memory> #include <memory>
#include <mutex> #include <mutex>
#include <optional> #include <optional>
@@ -182,6 +183,12 @@ struct RenderPass {
bool resolveNeedsConversion = false; bool resolveNeedsConversion = false;
bool resolveNeedsShaderSampling = false; bool resolveNeedsShaderSampling = false;
bool resolveLinearSampling = false; bool resolveLinearSampling = false;
bool displayCopyResolve = false;
// This pass is the continuation resolve_pass opened after a GXCopyDisp, so its
// clears describe that copy's EFB reset rather than anything the game drew.
// Deliberately not set for GXCopyTex: a mid-frame copy clear establishes the
// background the rest of the frame draws over, and an eye still needs it.
bool postCopyClear = false;
bool snapshotColorResolveSource = false; bool snapshotColorResolveSource = false;
bool efbTarget = false; bool efbTarget = false;
std::vector<tex_palette_conv::ConvRequest> paletteConvs; std::vector<tex_palette_conv::ConvRequest> paletteConvs;
@@ -189,6 +196,26 @@ struct RenderPass {
static std::vector<RenderPass> g_renderPasses; static std::vector<RenderPass> g_renderPasses;
static u32 g_currentRenderPass = UINT32_MAX; static u32 g_currentRenderPass = UINT32_MAX;
// Immersive-replay EFB controls. The settings overlay writes these from the UI
// thread while the frame worker reads them mid-encode, so they are atomic. Both
// default to the corrected behaviour; clearing either restores the raw replay
// for A/B comparison without a rebuild.
static std::atomic_bool g_stereoStopAtDisplayCopy{true};
static std::atomic_bool g_stereoSkipCopyClears{true};
void set_stereo_stop_at_display_copy(bool value) noexcept {
g_stereoStopAtDisplayCopy.store(value, std::memory_order_relaxed);
}
bool get_stereo_stop_at_display_copy() noexcept {
return g_stereoStopAtDisplayCopy.load(std::memory_order_relaxed);
}
void set_stereo_skip_copy_clears(bool value) noexcept {
g_stereoSkipCopyClears.store(value, std::memory_order_relaxed);
}
bool get_stereo_skip_copy_clears() noexcept {
return g_stereoSkipCopyClears.load(std::memory_order_relaxed);
}
// Recycle command storage: discarding passes used to free their command lists too, so each frame // Recycle command storage: discarding passes used to free their command lists too, so each frame
// rebuilt hundreds of KB from zero capacity. The passes themselves are cheap to recreate. // rebuilt hundreds of KB from zero capacity. The passes themselves are cheap to recreate.
using CommandListPool = std::vector<CommandList>; using CommandListPool = std::vector<CommandList>;
@@ -620,6 +647,10 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
.clearDepthValue = clearDepthValue, .clearDepthValue = clearDepthValue,
.clearColor = useAttachmentColorClear, .clearColor = useAttachmentColorClear,
.clearDepth = useAttachmentDepthClear, .clearDepth = useAttachmentDepthClear,
// This continuation still renders into the same main EFB attachments.
// Stereo replay filters on this flag; dropping it after a GX copy made
// every later race pass mono-only and left the eye targets cleared.
.efbTarget = prevPass.efbTarget,
}; };
push_render_pass(std::move(newPass)); push_render_pass(std::move(newPass));
++g_currentRenderPass; ++g_currentRenderPass;
@@ -643,6 +674,7 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
}, },
.depth = clearDepthValue, .depth = clearDepthValue,
.useScissor = !clearFullTarget, .useScissor = !clearFullTarget,
.copyClear = true,
.scissor = rect, .scissor = rect,
}); });
} }
@@ -650,6 +682,25 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
push_command(CommandType::SetScissor, Command::Data{.setScissor = g_cachedScissor}); push_command(CommandType::SetScissor, Command::Data{.setScissor = g_cachedScissor});
} }
void mark_last_resolve_as_display_copy() noexcept {
if (g_currentRenderPass == 0 || g_currentRenderPass > g_renderPasses.size()) {
Log.warn("Could not identify the render pass preceding a GX display copy");
return;
}
auto& resolvedPass = g_renderPasses[g_currentRenderPass - 1];
if (!resolvedPass.resolveTarget || !resolvedPass.efbTarget) {
Log.warn("GX display-copy marker did not follow a main-EFB resolve");
return;
}
resolvedPass.displayCopyResolve = true;
// resolve_pass has already opened the continuation that carries this copy's
// EFB reset, and only here is it known to belong to a display copy. The guard
// above admits g_currentRenderPass == size(), which has no continuation.
if (g_currentRenderPass < g_renderPasses.size()) {
g_renderPasses[g_currentRenderPass].postCopyClear = true;
}
}
void queue_palette_conv(tex_palette_conv::ConvRequest req) { void queue_palette_conv(tex_palette_conv::ConvRequest req) {
if (!has_current_render_pass()) { if (!has_current_render_pass()) {
Log.warn("Dropping palette conversion without an active render pass"); Log.warn("Dropping palette conversion without an active render pass");
@@ -1102,17 +1153,99 @@ void abort_frame() noexcept {
end_pipeline_frame(); end_pipeline_frame();
} }
static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame) noexcept { // What immersive replay should reproduce out of a sealed frame: the rectangle
size_t requiredBytes = 0; // the game presented, and the last pass that still contributes to it.
for (const auto& pass : g_renderPasses) { struct StereoDisplaySource {
if (!pass.efbTarget) { ClipRect region{};
// Inclusive index of the pass holding the final GXCopyDisp resolve. Passes
// after it only reset the EFB for the next frame, so an eye that replays them
// erases the image it just built. -1 means no display copy was found and the
// whole pass list is replayed against the full-EFB fallback region.
int32_t lastDisplayCopyPass = -1;
bool foundDisplayCopy = false;
};
static StereoDisplaySource stereo_display_source(const std::vector<RenderPass>& passes) noexcept {
StereoDisplaySource out{};
ClipRect fullRegion{};
for (const auto& pass : passes) {
if (!pass.efbTarget || pass.targetSize.width == 0 || pass.targetSize.height == 0) {
continue; continue;
} }
fullRegion = {
.x = 0,
.y = 0,
.width = static_cast<int32_t>(pass.targetSize.width),
.height = static_cast<int32_t>(pass.targetSize.height),
};
break;
}
// The desktop present source is replaced by each GXCopyDisp, so the final
// valid display copy—not a union of every copy—is the frame shown to users.
for (auto it = passes.rbegin(); it != passes.rend(); ++it) {
const auto& pass = *it;
if (!pass.efbTarget || !pass.displayCopyResolve || pass.targetSize.width == 0 ||
pass.targetSize.height == 0) {
continue;
}
const int32_t passIndex = static_cast<int32_t>(std::distance(passes.begin(), it.base())) - 1;
const int32_t targetWidth = static_cast<int32_t>(pass.targetSize.width);
const int32_t targetHeight = static_cast<int32_t>(pass.targetSize.height);
const int32_t left = std::clamp(pass.resolveRect.x, 0, targetWidth);
const int32_t top = std::clamp(pass.resolveRect.y, 0, targetHeight);
const int32_t right = std::clamp(pass.resolveRect.x + pass.resolveRect.width, left, targetWidth);
const int32_t bottom = std::clamp(pass.resolveRect.y + pass.resolveRect.height, top, targetHeight);
if (right <= left || bottom <= top) {
continue;
}
out.region = {left, top, right - left, bottom - top};
out.lastDisplayCopyPass = passIndex;
out.foundDisplayCopy = true;
return out;
}
out.region = fullRegion;
return out;
}
static void log_stereo_display_source_region(ClipRect region, bool foundDisplayCopy) noexcept {
static ClipRect lastLogged{};
static bool logged = false;
if (region.width > 0 && region.height > 0 && (!logged || region != lastLogged)) {
logged = true;
lastLogged = region;
Log.info("Immersive display-copy source region: {}x{} at ({}, {}){}", region.width,
region.height, region.x, region.y,
foundDisplayCopy ? "" : " (full-EFB fallback)");
}
}
static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame) noexcept {
const StereoDisplaySource displaySource = stereo_display_source(g_renderPasses);
const ClipRect displayRegion = displaySource.region;
// This is the producer-side preparation path; eye replay can query the pure
// helper concurrently without touching this diagnostic state.
log_stereo_display_source_region(displayRegion, displaySource.foundDisplayCopy);
size_t requiredBytes = 0;
size_t efbPassCount = 0;
size_t perspectiveDrawCount = 0;
size_t replayPerspectiveDrawCount = 0;
for (const auto& pass : g_renderPasses) {
if (pass.efbTarget) {
++efbPassCount;
}
for (const auto& command : pass.commands) { for (const auto& command : pass.commands) {
if (command.type != CommandType::Draw || command.data.draw.type != ShaderType::GX || if (command.type != CommandType::Draw || command.data.draw.type != ShaderType::GX ||
!command.data.draw.gx.uniformReplayLayout.perspective) { !command.data.draw.gx.uniformReplayLayout.perspective) {
continue; continue;
} }
++perspectiveDrawCount;
if (!pass.efbTarget) {
continue;
}
++replayPerspectiveDrawCount;
const auto& draw = command.data.draw.gx; const auto& draw = command.data.draw.gx;
const auto& layout = draw.uniformReplayLayout; const auto& layout = draw.uniformReplayLayout;
const size_t projectionEnd = static_cast<size_t>(layout.projectionOffset) + sizeof(Mat4x4<float>); const size_t projectionEnd = static_cast<size_t>(layout.projectionOffset) + sizeof(Mat4x4<float>);
@@ -1128,6 +1261,13 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame)
requiredBytes += static_cast<size_t>(draw.uniformRange.size) * AURORA_STEREO_EYE_COUNT; requiredBytes += static_cast<size_t>(draw.uniformRange.size) * AURORA_STEREO_EYE_COUNT;
} }
} }
static bool replayCoverageLogged = false;
if (!replayCoverageLogged) {
replayCoverageLogged = true;
Log.info("Immersive replay coverage: {} of {} passes target the EFB; {} of {} perspective draws replay",
efbPassCount, g_renderPasses.size(), replayPerspectiveDrawCount,
perspectiveDrawCount);
}
// end_batch_impl appends MaxUniformSize bytes after this for safe dynamic-offset reads. // end_batch_impl appends MaxUniformSize bytes after this for safe dynamic-offset reads.
if (requiredBytes > UniformBufferSize || if (requiredBytes > UniformBufferSize ||
@@ -1154,7 +1294,13 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame)
draw.stereoUniformRanges[eyeIndex] = range; draw.stereoUniformRanges[eyeIndex] = range;
const auto& eye = stereoFrame.eyes[eyeIndex]; const auto& eye = stereoFrame.eyes[eyeIndex];
std::memcpy(uniform.data() + layout.projectionOffset, &eye.projection, sizeof(eye.projection)); Mat4x4<float> gameProjection;
std::memcpy(&gameProjection, uniform.data() + layout.projectionOffset,
sizeof(gameProjection));
const auto projection =
stereo_replay::compose_projection(eye.projection, gameProjection);
std::memcpy(uniform.data() + layout.projectionOffset, &projection,
sizeof(projection));
for (uint32_t matrix = 0; matrix < layout.positionMatrixCount; ++matrix) { for (uint32_t matrix = 0; matrix < layout.positionMatrixCount; ++matrix) {
if ((layout.positionMatrixMask & (1u << matrix)) == 0) { if ((layout.positionMatrixMask & (1u << matrix)) == 0) {
@@ -1174,13 +1320,13 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame)
std::memcpy(uniform.data() + offset, &transformed, sizeof(transformed)); std::memcpy(uniform.data() + offset, &transformed, sizeof(transformed));
} }
if (pass.targetSize.width != 0 && pass.targetSize.height != 0) { if (displayRegion.width > 0 && displayRegion.height > 0) {
float renderSize[2]; float renderSize[2];
std::memcpy(renderSize, uniform.data() + 8, sizeof(renderSize)); std::memcpy(renderSize, uniform.data() + 8, sizeof(renderSize));
renderSize[0] *= static_cast<float>(eye.target.size.width) / renderSize[0] *= static_cast<float>(eye.target.size.width) /
static_cast<float>(pass.targetSize.width); static_cast<float>(displayRegion.width);
renderSize[1] *= static_cast<float>(eye.target.size.height) / renderSize[1] *= static_cast<float>(eye.target.size.height) /
static_cast<float>(pass.targetSize.height); static_cast<float>(displayRegion.height);
std::memcpy(uniform.data() + 8, renderSize, sizeof(renderSize)); std::memcpy(uniform.data() + 8, renderSize, sizeof(renderSize));
} }
} }
@@ -1296,8 +1442,12 @@ struct RenderInvocation {
int32_t interpolatedFrame = -1; int32_t interpolatedFrame = -1;
uint32_t stereoEye = UINT32_MAX; uint32_t stereoEye = UINT32_MAX;
const ReplayTarget* target = nullptr; const ReplayTarget* target = nullptr;
ClipRect replaySourceRegion{};
// Inclusive index of the last pass to replay; -1 replays every pass.
int32_t replayLastPass = -1;
bool finalize = true; bool finalize = true;
bool replayOnlyEfb = false; bool replayOnlyEfb = false;
bool skipCopyClears = false;
bool encodeTextureBakes = true; bool encodeTextureBakes = true;
bool encodeResolves = true; bool encodeResolves = true;
bool captureDepth = true; bool captureDepth = true;
@@ -1313,6 +1463,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
// interpolation weight, so encode them on the native render and let replay slots sample them. // interpolation weight, so encode them on the native render and let replay slots sample them.
for (u32 i = 0; i < renderPasses.size(); ++i) { for (u32 i = 0; i < renderPasses.size(); ++i) {
const auto& passInfo = renderPasses[i]; const auto& passInfo = renderPasses[i];
if (invocation.replayLastPass >= 0 && i > static_cast<u32>(invocation.replayLastPass)) {
// Only immersive replay sets this, and it never encodes bakes or resolves,
// so nothing later in the list is owed any work.
break;
}
if (invocation.replayOnlyEfb && !passInfo.efbTarget) { if (invocation.replayOnlyEfb && !passInfo.efbTarget) {
continue; continue;
} }
@@ -1331,6 +1486,10 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
} }
const bool overrideTarget = invocation.target != nullptr && passInfo.efbTarget; const bool overrideTarget = invocation.target != nullptr && passInfo.efbTarget;
// A GX copy clear resets the Wii's reused EFB for the next frame. An eye
// attachment is built fresh per frame and per eye, so reproducing that reset
// only erases what the replay already drew.
const bool dropCopyClear = invocation.skipCopyClears && overrideTarget && passInfo.postCopyClear;
const auto colorView = overrideTarget ? invocation.target->colorView : passInfo.colorView; const auto colorView = overrideTarget ? invocation.target->colorView : passInfo.colorView;
const auto resolveView = overrideTarget ? invocation.target->resolveView : passInfo.resolveView; const auto resolveView = overrideTarget ? invocation.target->resolveView : passInfo.resolveView;
const auto depthView = overrideTarget ? invocation.target->depthView : passInfo.depthView; const auto depthView = overrideTarget ? invocation.target->depthView : passInfo.depthView;
@@ -1338,7 +1497,7 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
wgpu::RenderPassColorAttachment{ wgpu::RenderPassColorAttachment{
.view = colorView, .view = colorView,
.resolveTarget = resolveView, .resolveTarget = resolveView,
.loadOp = passInfo.clearColor ? wgpu::LoadOp::Clear : wgpu::LoadOp::Load, .loadOp = passInfo.clearColor && !dropCopyClear ? wgpu::LoadOp::Clear : wgpu::LoadOp::Load,
.storeOp = wgpu::StoreOp::Store, .storeOp = wgpu::StoreOp::Store,
.clearValue = .clearValue =
{ {
@@ -1351,7 +1510,7 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
}; };
const wgpu::RenderPassDepthStencilAttachment depthStencilAttachment{ const wgpu::RenderPassDepthStencilAttachment depthStencilAttachment{
.view = depthView, .view = depthView,
.depthLoadOp = passInfo.clearDepth ? wgpu::LoadOp::Clear : wgpu::LoadOp::Load, .depthLoadOp = passInfo.clearDepth && !dropCopyClear ? wgpu::LoadOp::Clear : wgpu::LoadOp::Load,
.depthStoreOp = wgpu::StoreOp::Store, .depthStoreOp = wgpu::StoreOp::Store,
.depthClearValue = passInfo.clearDepthValue, .depthClearValue = passInfo.clearDepthValue,
}; };
@@ -1479,12 +1638,20 @@ void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedF
void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd,
const StereoReplayFrame& stereoFrame, uint32_t eye, bool finalize) { const StereoReplayFrame& stereoFrame, uint32_t eye, bool finalize) {
CHECK(eye < AURORA_STEREO_EYE_COUNT, "invalid stereo eye {}", eye); CHECK(eye < AURORA_STEREO_EYE_COUNT, "invalid stereo eye {}", eye);
const auto displaySource = stereo_display_source(frame.data().passes);
// GXCopyDisp publishes the frame and then clears the EFB for the next one.
// The eye is a fresh per-frame attachment, not the reused EFB, so replaying
// past that copy blanks the very image the game presented.
const int32_t lastPass = get_stereo_stop_at_display_copy() ? displaySource.lastDisplayCopyPass : -1;
render_impl(frame.data().passes, cmd, render_impl(frame.data().passes, cmd,
RenderInvocation{ RenderInvocation{
.stereoEye = eye, .stereoEye = eye,
.target = &stereoFrame.eyes[eye].target, .target = &stereoFrame.eyes[eye].target,
.replaySourceRegion = displaySource.region,
.replayLastPass = lastPass,
.finalize = finalize, .finalize = finalize,
.replayOnlyEfb = true, .replayOnlyEfb = true,
.skipCopyClears = get_stereo_skip_copy_clears(),
.encodeTextureBakes = false, .encodeTextureBakes = false,
.encodeResolves = false, .encodeResolves = false,
.captureDepth = false, .captureDepth = false,
@@ -1530,11 +1697,33 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
const auto& sourceSize = renderPasses[idx].targetSize; const auto& sourceSize = renderPasses[idx].targetSize;
const bool overrideTarget = invocation.target != nullptr && renderPasses[idx].efbTarget; const bool overrideTarget = invocation.target != nullptr && renderPasses[idx].efbTarget;
const auto targetSize = overrideTarget ? invocation.target->size : sourceSize; const auto targetSize = overrideTarget ? invocation.target->size : sourceSize;
const float scaleX = overrideTarget && sourceSize.width != 0 const int32_t sourceWidth = static_cast<int32_t>(sourceSize.width);
? static_cast<float>(targetSize.width) / static_cast<float>(sourceSize.width) const int32_t sourceHeight = static_cast<int32_t>(sourceSize.height);
int32_t sourceRegionLeft = 0;
int32_t sourceRegionTop = 0;
int32_t sourceRegionRight = sourceWidth;
int32_t sourceRegionBottom = sourceHeight;
if (overrideTarget && invocation.replaySourceRegion.width > 0 &&
invocation.replaySourceRegion.height > 0) {
const auto& region = invocation.replaySourceRegion;
sourceRegionLeft = std::clamp(region.x, 0, sourceWidth);
sourceRegionTop = std::clamp(region.y, 0, sourceHeight);
sourceRegionRight = std::clamp(region.x + region.width, sourceRegionLeft, sourceWidth);
sourceRegionBottom = std::clamp(region.y + region.height, sourceRegionTop, sourceHeight);
if (sourceRegionRight == sourceRegionLeft || sourceRegionBottom == sourceRegionTop) {
sourceRegionLeft = 0;
sourceRegionTop = 0;
sourceRegionRight = sourceWidth;
sourceRegionBottom = sourceHeight;
}
}
const int32_t sourceRegionWidth = sourceRegionRight - sourceRegionLeft;
const int32_t sourceRegionHeight = sourceRegionBottom - sourceRegionTop;
const float scaleX = overrideTarget && sourceRegionWidth > 0
? static_cast<float>(targetSize.width) / static_cast<float>(sourceRegionWidth)
: 1.0f; : 1.0f;
const float scaleY = overrideTarget && sourceSize.height != 0 const float scaleY = overrideTarget && sourceRegionHeight > 0
? static_cast<float>(targetSize.height) / static_cast<float>(sourceSize.height) ? static_cast<float>(targetSize.height) / static_cast<float>(sourceRegionHeight)
: 1.0f; : 1.0f;
for (const auto& cmd : renderPasses[idx].commands) { for (const auto& cmd : renderPasses[idx].commands) {
@@ -1563,28 +1752,29 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
// reproduced in clip space. Passing the raw swapped pair diverged per backend in release builds. // reproduced in clip space. Passing the raw swapped pair diverged per backend in release builds.
const float minDepth = std::clamp(std::min(vp.znear, vp.zfar), 0.0f, 1.0f); const float minDepth = std::clamp(std::min(vp.znear, vp.zfar), 0.0f, 1.0f);
const float maxDepth = std::clamp(std::max(vp.znear, vp.zfar), 0.0f, 1.0f); const float maxDepth = std::clamp(std::max(vp.znear, vp.zfar), 0.0f, 1.0f);
pass.SetViewport(vp.left * scaleX, vp.top * scaleY, vp.width * scaleX, vp.height * scaleY, pass.SetViewport((vp.left - static_cast<float>(sourceRegionLeft)) * scaleX,
minDepth, maxDepth); (vp.top - static_cast<float>(sourceRegionTop)) * scaleY,
vp.width * scaleX, vp.height * scaleY, minDepth, maxDepth);
} break; } break;
case CommandType::SetScissor: { case CommandType::SetScissor: {
const auto& sc = cmd.data.setScissor; const auto& sc = cmd.data.setScissor;
const auto sourceLeft = std::clamp(sc.x, 0, static_cast<int32_t>(sourceSize.width)); const auto sourceLeft = std::clamp(sc.x, sourceRegionLeft, sourceRegionRight);
const auto sourceTop = std::clamp(sc.y, 0, static_cast<int32_t>(sourceSize.height)); const auto sourceTop = std::clamp(sc.y, sourceRegionTop, sourceRegionBottom);
const auto sourceRight = const auto sourceRight =
std::clamp(sc.x + sc.width, sourceLeft, static_cast<int32_t>(sourceSize.width)); std::clamp(sc.x + sc.width, sourceLeft, sourceRegionRight);
const auto sourceBottom = const auto sourceBottom =
std::clamp(sc.y + sc.height, sourceTop, static_cast<int32_t>(sourceSize.height)); std::clamp(sc.y + sc.height, sourceTop, sourceRegionBottom);
const auto left = static_cast<uint32_t>(std::clamp( const auto left = static_cast<uint32_t>(std::clamp(
static_cast<int32_t>(std::floor(static_cast<float>(sourceLeft) * scaleX)), 0, static_cast<int32_t>(std::floor(static_cast<float>(sourceLeft - sourceRegionLeft) * scaleX)), 0,
static_cast<int32_t>(targetSize.width))); static_cast<int32_t>(targetSize.width)));
const auto top = static_cast<uint32_t>(std::clamp( const auto top = static_cast<uint32_t>(std::clamp(
static_cast<int32_t>(std::floor(static_cast<float>(sourceTop) * scaleY)), 0, static_cast<int32_t>(std::floor(static_cast<float>(sourceTop - sourceRegionTop) * scaleY)), 0,
static_cast<int32_t>(targetSize.height))); static_cast<int32_t>(targetSize.height)));
const auto right = static_cast<uint32_t>(std::clamp( const auto right = static_cast<uint32_t>(std::clamp(
static_cast<int32_t>(std::ceil(static_cast<float>(sourceRight) * scaleX)), static_cast<int32_t>(std::ceil(static_cast<float>(sourceRight - sourceRegionLeft) * scaleX)),
static_cast<int32_t>(left), static_cast<int32_t>(targetSize.width))); static_cast<int32_t>(left), static_cast<int32_t>(targetSize.width)));
const auto bottom = static_cast<uint32_t>(std::clamp( const auto bottom = static_cast<uint32_t>(std::clamp(
static_cast<int32_t>(std::ceil(static_cast<float>(sourceBottom) * scaleY)), static_cast<int32_t>(std::ceil(static_cast<float>(sourceBottom - sourceRegionTop) * scaleY)),
static_cast<int32_t>(top), static_cast<int32_t>(targetSize.height))); static_cast<int32_t>(top), static_cast<int32_t>(targetSize.height)));
pass.SetScissorRect(left, top, right - left, bottom - top); pass.SetScissorRect(left, top, right - left, bottom - top);
} break; } break;
@@ -1604,9 +1794,53 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
} }
gx::render(draw.gx, pass, encodeState, renderPasses[idx].requireReadyPipelines, uniformOverride); gx::render(draw.gx, pass, encodeState, renderPasses[idx].requireReadyPipelines, uniformOverride);
} break; } break;
case ShaderType::Clear: case ShaderType::Clear: {
clear::render(draw.clear, pass, renderPasses[idx].targetSize, encodeState.currentPipeline); auto clearDraw = draw.clear;
break; if (invocation.skipCopyClears && overrideTarget && clearDraw.copyClear &&
renderPasses[idx].postCopyClear) {
// The scissored twin of the attachment-load-op case above: the copy's
// EFB reset, rescaled into eye space, covers the whole eye.
break;
}
if (overrideTarget && clearDraw.useScissor) {
const auto& sc = clearDraw.scissor;
const auto sourceLeft = std::clamp(sc.x, sourceRegionLeft, sourceRegionRight);
const auto sourceTop = std::clamp(sc.y, sourceRegionTop, sourceRegionBottom);
const auto sourceRight =
std::clamp(sc.x + sc.width, sourceLeft, sourceRegionRight);
const auto sourceBottom =
std::clamp(sc.y + sc.height, sourceTop, sourceRegionBottom);
const auto left = std::clamp(
static_cast<int32_t>(
std::floor(static_cast<float>(sourceLeft - sourceRegionLeft) * scaleX)),
0,
static_cast<int32_t>(targetSize.width));
const auto top = std::clamp(
static_cast<int32_t>(
std::floor(static_cast<float>(sourceTop - sourceRegionTop) * scaleY)),
0,
static_cast<int32_t>(targetSize.height));
const auto right = std::clamp(
static_cast<int32_t>(
std::ceil(static_cast<float>(sourceRight - sourceRegionLeft) * scaleX)),
left,
static_cast<int32_t>(targetSize.width));
const auto bottom = std::clamp(
static_cast<int32_t>(
std::ceil(static_cast<float>(sourceBottom - sourceRegionTop) * scaleY)),
top,
static_cast<int32_t>(targetSize.height));
clearDraw.scissor = ClipRect{
.x = left,
.y = top,
.width = right - left,
.height = bottom - top,
};
}
// Clear draws set their own viewport and scissor. Stereo replay must use
// the eye attachment extent here, not the original EFB/desktop extent.
clear::render(clearDraw, pass, targetSize, encodeState.currentPipeline);
} break;
} }
} break; } break;
case CommandType::DebugMarker: { case CommandType::DebugMarker: {
+11
View File
@@ -362,6 +362,17 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
const std::array<u32, 3>* copyFilterCoefficients = nullptr, bool forceOpaqueAlpha = false, const std::array<u32, 3>* copyFilterCoefficients = nullptr, bool forceOpaqueAlpha = false,
float copyFilterRowStride = 1.0f, bool clampTop = false, bool clampBottom = false, float copyFilterRowStride = 1.0f, bool clampTop = false, bool clampBottom = false,
bool persistentCopy = false); bool persistentCopy = false);
// Marks the resolve immediately preceding the current continuation pass as the
// EFB-to-display copy. Immersive replay uses its source rectangle as the eye
// viewport instead of exposing the Wii's larger scratch EFB workspace.
void mark_last_resolve_as_display_copy() noexcept;
// Immersive-replay EFB controls, both on by default. Safe to flip at any time:
// the frame worker reads them atomically once per eye.
void set_stereo_stop_at_display_copy(bool value) noexcept;
bool get_stereo_stop_at_display_copy() noexcept;
void set_stereo_skip_copy_clears(bool value) noexcept;
bool get_stereo_skip_copy_clears() noexcept;
void begin_offscreen(uint32_t width, uint32_t height); void begin_offscreen(uint32_t width, uint32_t height);
void end_offscreen(); void end_offscreen();
+16
View File
@@ -4,6 +4,22 @@
namespace aurora::gfx::stereo_replay { namespace aurora::gfx::stereo_replay {
// An OpenXR eye supplies the shape of its asymmetric frustum, but the sealed
// GX draw already contains the depth mapping adjusted for that draw's GX
// viewport and Aurora's reversed-Z convention. Replacing the complete matrix
// would pair an unrelated depth range with the original pipeline compare and
// clear state, which can reject the entire eye. Replace only the four
// perspective-frustum coefficients and preserve every depth-related element.
inline Mat4x4<float> compose_projection(const Mat4x4<float>& eyeFrustum,
const Mat4x4<float>& gameProjection) noexcept {
Mat4x4<float> out = gameProjection;
out.m0[0] = eyeFrustum.m0[0];
out.m0[2] = eyeFrustum.m0[2];
out.m1[1] = eyeFrustum.m1[1];
out.m1[2] = eyeFrustum.m1[2];
return out;
}
// Aurora stores the GX 3x4 matrices row-major. The vertex shader consumes // Aurora stores the GX 3x4 matrices row-major. The vertex shader consumes
// them as vec4 * mat3x4, which is equivalent to the original column-vector // them as vec4 * mat3x4, which is equivalent to the original column-vector
// affine transform. Applying an eye-space delta therefore composes delta * // affine transform. Applying an eye-space delta therefore composes delta *
+1
View File
@@ -132,6 +132,7 @@ extern char g_gameName[4];
void wait_for_frame_worker() noexcept; void wait_for_frame_worker() noexcept;
std::chrono::nanoseconds wait_for_frame_worker_sealed() noexcept; std::chrono::nanoseconds wait_for_frame_worker_sealed() noexcept;
bool wait_for_frame_worker_for(std::chrono::microseconds timeout) noexcept; bool wait_for_frame_worker_for(std::chrono::microseconds timeout) noexcept;
void quiesce_frame_worker() noexcept;
std::recursive_mutex& renderer_gpu_mutex() noexcept; std::recursive_mutex& renderer_gpu_mutex() noexcept;
template <typename T> template <typename T>
+11 -2
View File
@@ -19,16 +19,25 @@ struct EyeImage {
struct SinkFrame { struct SinkFrame {
uint64_t frameToken = 0; uint64_t frameToken = 0;
uint32_t logicalFrame = 0; uint32_t logicalFrame = 0;
AuroraStereoFrameMode mode = AURORA_STEREO_FRAME_IMMERSIVE_REPLAY;
std::array<EyeImage, AURORA_STEREO_EYE_COUNT> eyes{}; std::array<EyeImage, AURORA_STEREO_EYE_COUNT> eyes{};
}; };
// Runs synchronously on the frame worker after both eye replays have been // Runs synchronously on the frame worker after both eye replays have been
// encoded and before its command buffer is submitted. Backend interop code // encoded and before its command buffer is submitted. Backend interop code
// can append copies/import transitions to the same encoder here. // can append copies/import transitions to the same encoder here.
using SinkCallback = void (*)(wgpu::CommandEncoder& encoder, const SinkFrame& frame, void* userdata) noexcept; using SinkCallback = bool (*)(wgpu::CommandEncoder& encoder, const SinkFrame& frame, void* userdata) noexcept;
// Called immediately after the command buffer containing the sink's encoded
// work is submitted to Dawn's queue, while Aurora's queue-submit mutex remains
// held. It may enqueue native work on that same queue, but must not wait or
// call back into Aurora. This is the handoff point for backend-native follow-up
// work; its completion notification may wake the owner thread to release the
// acquired OpenXR images and end the frame.
using SubmitCallback = void (*)(const SinkFrame& frame, void* userdata) noexcept;
// Internal Aurora hook: D3D12/Vulkan OpenXR interop owns this registration. // Internal Aurora hook: D3D12/Vulkan OpenXR interop owns this registration.
// Registration changes must happen while the frame worker is idle. // Registration changes must happen while the frame worker is idle.
void set_sink(SinkCallback callback, void* userdata) noexcept; void set_sink(SinkCallback callback, SubmitCallback submitted, void* userdata) noexcept;
inline void set_sink(SinkCallback callback, void* userdata) noexcept { set_sink(callback, nullptr, userdata); }
} // namespace aurora::stereo } // namespace aurora::stereo
+751
View File
@@ -0,0 +1,751 @@
#include <aurora/d3d12_interop.h>
#include "../internal.hpp"
#include "../stereo.hpp"
#include "gpu.hpp"
#if defined(_WIN32) && defined(WEBGPU_DAWN) && defined(DAWN_ENABLE_BACKEND_D3D12)
#include <dawn/native/D3D12Backend.h>
#include <d3d12.h>
#include <dxgi1_4.h>
#include <windows.h>
#include <wrl/client.h>
#include <algorithm>
#include <array>
#include <chrono>
#include <cstdint>
#include <limits>
#include <memory>
#include <mutex>
#include <utility>
#include <vector>
namespace aurora::d3d12_interop {
namespace {
using Microsoft::WRL::ComPtr;
Module Log("aurora::d3d12_interop");
constexpr uint64_t kUnfencedSubmission = (std::numeric_limits<uint64_t>::max)();
constexpr char kGetDeviceExport[] =
"?GetD3D12Device@d3d12@native@dawn@@YA?AV?$ComPtr@UID3D12Device@@@WRL@Microsoft@@PEAUWGPUDeviceImpl@@@Z";
constexpr char kGetQueueExport[] =
"?GetD3D12CommandQueue@d3d12@native@dawn@@YA?AV?$ComPtr@UID3D12CommandQueue@@@WRL@Microsoft@@PEAUWGPUDeviceImpl@@@Z";
struct NativeObjects {
ComPtr<ID3D12Device> device;
ComPtr<ID3D12CommandQueue> queue;
};
int64_t to_dxgi_format(wgpu::TextureFormat format) noexcept {
switch (format) {
case wgpu::TextureFormat::RGBA8Unorm:
return DXGI_FORMAT_R8G8B8A8_UNORM;
case wgpu::TextureFormat::RGBA8UnormSrgb:
return DXGI_FORMAT_R8G8B8A8_UNORM_SRGB;
case wgpu::TextureFormat::BGRA8Unorm:
return DXGI_FORMAT_B8G8R8A8_UNORM;
case wgpu::TextureFormat::BGRA8UnormSrgb:
return DXGI_FORMAT_B8G8R8A8_UNORM_SRGB;
case wgpu::TextureFormat::RGBA16Float:
return DXGI_FORMAT_R16G16B16A16_FLOAT;
default:
return DXGI_FORMAT_UNKNOWN;
}
}
bool same_copy_family(DXGI_FORMAT left, DXGI_FORMAT right) noexcept {
const auto family = [](DXGI_FORMAT format) {
switch (format) {
case DXGI_FORMAT_R8G8B8A8_TYPELESS:
case DXGI_FORMAT_R8G8B8A8_UNORM:
case DXGI_FORMAT_R8G8B8A8_UNORM_SRGB:
return 1;
case DXGI_FORMAT_B8G8R8A8_TYPELESS:
case DXGI_FORMAT_B8G8R8A8_UNORM:
case DXGI_FORMAT_B8G8R8A8_UNORM_SRGB:
return 2;
case DXGI_FORMAT_R16G16B16A16_TYPELESS:
case DXGI_FORMAT_R16G16B16A16_FLOAT:
return 3;
default:
return 0;
}
};
const int leftFamily = family(left);
return leftFamily != 0 && leftFamily == family(right);
}
bool get_native_objects(NativeObjects& objects) noexcept {
if (!webgpu::g_device || webgpu::g_backendType != wgpu::BackendType::D3D12) {
return false;
}
#if defined(__MINGW32__)
// The distributed Dawn DLL is built by MSVC. LLVM-MinGW uses a different
// C++ symbol spelling, but Win64's calling ABI is identical. Resolve the two
// pinned native exports explicitly and keep all public interop C-compatible.
static_assert(sizeof(ComPtr<ID3D12Device>) == sizeof(void*));
HMODULE dawnModule = GetModuleHandleW(L"webgpu_dawn.dll");
if (dawnModule == nullptr) {
Log.error("webgpu_dawn.dll is not loaded; native D3D12 interop is unavailable");
return false;
}
using GetDeviceFn = ComPtr<ID3D12Device> (*)(WGPUDevice);
using GetQueueFn = ComPtr<ID3D12CommandQueue> (*)(WGPUDevice);
const auto getDevice = reinterpret_cast<GetDeviceFn>(GetProcAddress(dawnModule, kGetDeviceExport));
const auto getQueue = reinterpret_cast<GetQueueFn>(GetProcAddress(dawnModule, kGetQueueExport));
if (getDevice == nullptr || getQueue == nullptr) {
Log.error("Pinned Dawn native D3D12 exports are unavailable");
return false;
}
objects.device = getDevice(webgpu::g_device.Get());
objects.queue = getQueue(webgpu::g_device.Get());
#else
objects.device = dawn::native::d3d12::GetD3D12Device(webgpu::g_device.Get());
objects.queue = dawn::native::d3d12::GetD3D12CommandQueue(webgpu::g_device.Get());
#endif
return objects.device != nullptr && objects.queue != nullptr;
}
// Dawn exposes this descriptor only through a native C++ type. Its ABI is a
// chained header followed by a ComPtr, so spell the wire layout locally and
// enter Dawn through the ordinary WebGPU C API instead of linking a C++ ctor.
struct SharedTextureMemoryD3D12ResourceWire {
wgpu::ChainedStruct chain{};
ComPtr<ID3D12Resource> resource;
};
struct SharedFenceDxgiHandleWire {
wgpu::ChainedStruct chain{};
void* handle = nullptr;
};
struct IntermediateEye {
ComPtr<ID3D12Resource> resource;
wgpu::SharedTextureMemory memory;
wgpu::Texture texture;
wgpu::TextureFormat webgpuFormat = wgpu::TextureFormat::Undefined;
DXGI_FORMAT dxgiFormat = DXGI_FORMAT_UNKNOWN;
uint32_t width = 0;
uint32_t height = 0;
bool initialized = false;
bool accessBegun = false;
};
struct PendingTarget {
ComPtr<ID3D12Resource> resource;
uint32_t width = 0;
uint32_t height = 0;
DXGI_FORMAT format = DXGI_FORMAT_UNKNOWN;
};
struct InFlightCommand {
uint64_t fenceValue = 0;
ComPtr<ID3D12CommandAllocator> allocator;
ComPtr<ID3D12GraphicsCommandList> list;
// D3D12 command lists do not retain application resource references. Keep
// both sides of every copy alive until this submission's fence completes;
// an eye-size change may otherwise replace the bridge intermediate while
// the GPU is still reading it.
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> sources;
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> destinations;
};
class StereoBridge final {
public:
StereoBridge(NativeObjects objects, AuroraD3D12StereoSubmittedCallback callback,
void* userdata) noexcept
: m_device(std::move(objects.device)), m_queue(std::move(objects.queue)),
m_callback(callback), m_userdata(userdata) {}
bool Initialize() noexcept {
if (!webgpu::g_device.HasFeature(wgpu::FeatureName::SharedTextureMemoryD3D12Resource)) {
Log.error("Dawn device lacks SharedTextureMemoryD3D12Resource");
return false;
}
if (FAILED(m_device->CreateFence(0, D3D12_FENCE_FLAG_SHARED, IID_PPV_ARGS(&m_fence)))) {
Log.error("Could not create the D3D12 interop fence");
return false;
}
if (webgpu::g_device.HasFeature(wgpu::FeatureName::SharedFenceDXGISharedHandle)) {
HANDLE handle = nullptr;
if (SUCCEEDED(m_device->CreateSharedHandle(m_fence.Get(), nullptr, GENERIC_ALL, nullptr,
&handle))) {
SharedFenceDxgiHandleWire wire{};
wire.chain.sType = wgpu::SType::SharedFenceDXGISharedHandleDescriptor;
wire.handle = handle;
const wgpu::SharedFenceDescriptor descriptor{
.nextInChain = &wire.chain,
.label = "Aurora D3D12 stereo interop fence",
};
m_webgpuFence = webgpu::g_device.ImportSharedFence(&descriptor);
CloseHandle(handle);
}
}
if (!m_webgpuFence) {
// This is still ordered correctly because both APIs submit to the exact
// same D3D12 queue. The explicit shared fence additionally describes the
// dependency to Dawn when that optional feature is available.
Log.warn("Dawn shared-fence import is unavailable; using same-queue ordering");
}
return true;
}
~StereoBridge() {
if (!m_gpuIdle) {
(void)WaitForGpuLocked();
}
}
bool PrepareForDestruction() noexcept {
std::lock_guard lock(m_mutex);
return WaitForGpuLocked();
}
bool SetTargets(uint64_t token, const AuroraD3D12StereoTarget* targets,
uint32_t targetCount) noexcept {
if (token == 0 || targets == nullptr || targetCount == 0 ||
targetCount > AURORA_D3D12_STEREO_MAX_TARGETS) {
return false;
}
std::lock_guard lock(m_mutex);
if (m_framePending || m_encoded) {
return false;
}
for (uint32_t eye = 0; eye < targetCount; ++eye) {
if (targets[eye].resource == nullptr || targets[eye].width == 0 ||
targets[eye].height == 0 || targets[eye].dxgiFormat == DXGI_FORMAT_UNKNOWN) {
return false;
}
auto* resource = static_cast<ID3D12Resource*>(targets[eye].resource);
const D3D12_RESOURCE_DESC desc = resource->GetDesc();
if (desc.Dimension != D3D12_RESOURCE_DIMENSION_TEXTURE2D ||
desc.Width < targets[eye].width || desc.Height < targets[eye].height ||
desc.DepthOrArraySize != 1 || desc.MipLevels != 1 || desc.SampleDesc.Count != 1 ||
!same_copy_family(desc.Format, static_cast<DXGI_FORMAT>(targets[eye].dxgiFormat))) {
return false;
}
m_targets[eye] = {
.resource = resource,
.width = targets[eye].width,
.height = targets[eye].height,
.format = static_cast<DXGI_FORMAT>(targets[eye].dxgiFormat),
};
}
for (uint32_t eye = targetCount; eye < m_targets.size(); ++eye) {
m_targets[eye] = {};
}
m_frameToken = token;
m_targetCount = targetCount;
m_framePending = true;
return true;
}
bool Encode(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) noexcept {
std::lock_guard lock(m_mutex);
if (!m_framePending || m_encoded || frame.frameToken != m_frameToken) {
return false;
}
if (EncodeLocked(encoder, frame)) {
m_encoded = true;
return true;
}
PublishAndClearFrameLocked(frame.frameToken, false);
return false;
}
void Submitted(const stereo::SinkFrame& frame) noexcept {
std::lock_guard lock(m_mutex);
if (!m_framePending || !m_encoded || frame.frameToken != m_frameToken) {
return;
}
const bool success = EndAccessLocked() && EnqueueNativeCopyLocked();
PublishAndClearFrameLocked(frame.frameToken, success);
}
void CancelPending() noexcept {
uint64_t token = 0;
{
std::lock_guard lock(m_mutex);
if (!m_framePending) {
return;
}
token = m_frameToken;
if (m_encoded) {
EndAccessLocked();
}
PublishAndClearFrameLocked(token, false);
}
}
bool CancelBeforeEncode(uint64_t token) noexcept {
// Never make the XR pacing thread wait behind an in-progress Encode. A
// failed try-lock means Aurora may already own GPU-relevant work, so the
// submitted callback remains authoritative.
std::unique_lock lock(m_mutex, std::try_to_lock);
if (!lock.owns_lock()) {
return false;
}
if (token == 0 || !m_framePending || m_encoded || token != m_frameToken) {
return false;
}
ClearFrameLocked();
return true;
}
private:
bool EnsureIntermediate(uint32_t eye, const stereo::EyeImage& source) noexcept {
auto& intermediate = m_intermediates[eye];
const DXGI_FORMAT sourceFormat = static_cast<DXGI_FORMAT>(to_dxgi_format(source.format));
if (source.texture == nullptr || sourceFormat == DXGI_FORMAT_UNKNOWN ||
source.size.width != m_targets[eye].width || source.size.height != m_targets[eye].height ||
!same_copy_family(sourceFormat, m_targets[eye].format)) {
Log.error("Stereo eye {} does not match its OpenXR D3D12 target", eye);
return false;
}
if (intermediate.texture && intermediate.width == source.size.width &&
intermediate.height == source.size.height && intermediate.webgpuFormat == source.format) {
return true;
}
if (intermediate.accessBegun) {
return false;
}
intermediate = {};
const D3D12_HEAP_PROPERTIES heap{
.Type = D3D12_HEAP_TYPE_DEFAULT,
.CPUPageProperty = D3D12_CPU_PAGE_PROPERTY_UNKNOWN,
.MemoryPoolPreference = D3D12_MEMORY_POOL_UNKNOWN,
.CreationNodeMask = 1,
.VisibleNodeMask = 1,
};
const D3D12_RESOURCE_DESC resourceDescriptor{
.Dimension = D3D12_RESOURCE_DIMENSION_TEXTURE2D,
.Alignment = 0,
.Width = source.size.width,
.Height = source.size.height,
.DepthOrArraySize = 1,
.MipLevels = 1,
.Format = sourceFormat,
.SampleDesc = {1, 0},
.Layout = D3D12_TEXTURE_LAYOUT_UNKNOWN,
.Flags = D3D12_RESOURCE_FLAG_ALLOW_SIMULTANEOUS_ACCESS,
};
if (FAILED(m_device->CreateCommittedResource(
&heap, D3D12_HEAP_FLAG_NONE, &resourceDescriptor, D3D12_RESOURCE_STATE_COMMON,
nullptr, IID_PPV_ARGS(&intermediate.resource)))) {
Log.error("Could not create D3D12 stereo intermediate for eye {}", eye);
return false;
}
SharedTextureMemoryD3D12ResourceWire wire{};
wire.chain.sType = wgpu::SType::SharedTextureMemoryD3D12ResourceDescriptor;
wire.resource = intermediate.resource;
const wgpu::SharedTextureMemoryDescriptor memoryDescriptor{
.nextInChain = &wire.chain,
.label = eye == 0 ? "OpenXR left eye intermediate" : "OpenXR right eye intermediate",
};
intermediate.memory = webgpu::g_device.ImportSharedTextureMemory(&memoryDescriptor);
if (!intermediate.memory) {
Log.error("Dawn rejected D3D12 stereo intermediate for eye {}", eye);
intermediate = {};
return false;
}
wgpu::SharedTextureMemoryProperties properties{};
if (intermediate.memory.GetProperties(&properties) != wgpu::Status::Success ||
properties.size.width != source.size.width || properties.size.height != source.size.height ||
properties.format != source.format ||
(properties.usage & wgpu::TextureUsage::CopyDst) == wgpu::TextureUsage::None) {
Log.error("Dawn reported incompatible D3D12 shared-texture properties for eye {}", eye);
intermediate = {};
return false;
}
const wgpu::TextureDescriptor textureDescriptor{
.label = eye == 0 ? "OpenXR left eye shared texture" : "OpenXR right eye shared texture",
.usage = wgpu::TextureUsage::CopyDst,
.dimension = wgpu::TextureDimension::e2D,
.size = {source.size.width, source.size.height, 1},
.format = source.format,
.mipLevelCount = 1,
.sampleCount = 1,
};
intermediate.texture = intermediate.memory.CreateTexture(&textureDescriptor);
if (!intermediate.texture) {
Log.error("Dawn could not wrap D3D12 stereo intermediate for eye {}", eye);
intermediate = {};
return false;
}
intermediate.webgpuFormat = source.format;
intermediate.dxgiFormat = sourceFormat;
intermediate.width = source.size.width;
intermediate.height = source.size.height;
return true;
}
bool EncodeLocked(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) noexcept {
CollectCompletedCommandsLocked();
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
if (!EnsureIntermediate(eye, frame.eyes[eye])) {
return false;
}
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
auto& intermediate = m_intermediates[eye];
const std::array fences{m_webgpuFence};
const std::array values{m_lastExternalFenceValue};
wgpu::SharedTextureMemoryBeginAccessDescriptor begin{};
begin.initialized = intermediate.initialized;
if (m_webgpuFence && m_lastExternalFenceValue != 0) {
begin.fenceCount = 1;
begin.fences = fences.data();
begin.signaledValueCount = 1;
begin.signaledValues = values.data();
}
if (intermediate.memory.BeginAccess(intermediate.texture, &begin) != wgpu::Status::Success) {
Log.error("Dawn BeginAccess failed for stereo eye {}", eye);
for (uint32_t begunEye = 0; begunEye < eye; ++begunEye) {
wgpu::SharedTextureMemoryEndAccessState end{};
m_intermediates[begunEye].memory.EndAccess(m_intermediates[begunEye].texture, &end);
m_intermediates[begunEye].initialized = end.initialized;
m_intermediates[begunEye].accessBegun = false;
}
return false;
}
intermediate.accessBegun = true;
}
// Acquire every shared texture before recording any command that refers
// to one. If a later BeginAccess fails, the rollback above can therefore
// end the earlier accesses without leaving an unsubmitted copy that uses
// a texture after its access interval.
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
const auto& intermediate = m_intermediates[eye];
const wgpu::TexelCopyTextureInfo source{
.texture = *frame.eyes[eye].texture,
.mipLevel = 0,
.origin = {},
.aspect = wgpu::TextureAspect::All,
};
const wgpu::TexelCopyTextureInfo destination{
.texture = intermediate.texture,
.mipLevel = 0,
.origin = {},
.aspect = wgpu::TextureAspect::All,
};
const wgpu::Extent3D extent{intermediate.width, intermediate.height, 1};
encoder.CopyTextureToTexture(&source, &destination, &extent);
}
return true;
}
bool EndAccessLocked() noexcept {
bool success = true;
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
auto& intermediate = m_intermediates[eye];
if (!intermediate.accessBegun) {
success = false;
continue;
}
wgpu::SharedTextureMemoryEndAccessState end{};
if (intermediate.memory.EndAccess(intermediate.texture, &end) != wgpu::Status::Success) {
Log.error("Dawn EndAccess failed for stereo eye {}", eye);
success = false;
} else {
intermediate.initialized = end.initialized;
}
intermediate.accessBegun = false;
}
return success;
}
bool EnqueueNativeCopyLocked() noexcept {
ComPtr<ID3D12CommandAllocator> allocator;
ComPtr<ID3D12GraphicsCommandList> list;
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> sources;
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> destinations;
if (FAILED(m_device->CreateCommandAllocator(D3D12_COMMAND_LIST_TYPE_DIRECT,
IID_PPV_ARGS(&allocator))) ||
FAILED(m_device->CreateCommandList(0, D3D12_COMMAND_LIST_TYPE_DIRECT, allocator.Get(),
nullptr, IID_PPV_ARGS(&list)))) {
Log.error("Could not create the D3D12 stereo copy command list");
return false;
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
const auto& source = m_intermediates[eye];
const auto& destination = m_targets[eye];
sources[eye] = source.resource;
destinations[eye] = destination.resource;
const std::array barriers{
D3D12_RESOURCE_BARRIER{
.Type = D3D12_RESOURCE_BARRIER_TYPE_TRANSITION,
.Flags = D3D12_RESOURCE_BARRIER_FLAG_NONE,
.Transition = {source.resource.Get(), D3D12_RESOURCE_BARRIER_ALL_SUBRESOURCES,
D3D12_RESOURCE_STATE_COMMON, D3D12_RESOURCE_STATE_COPY_SOURCE},
},
D3D12_RESOURCE_BARRIER{
.Type = D3D12_RESOURCE_BARRIER_TYPE_TRANSITION,
.Flags = D3D12_RESOURCE_BARRIER_FLAG_NONE,
.Transition = {destination.resource.Get(), D3D12_RESOURCE_BARRIER_ALL_SUBRESOURCES,
D3D12_RESOURCE_STATE_RENDER_TARGET,
D3D12_RESOURCE_STATE_COPY_DEST},
},
};
list->ResourceBarrier(static_cast<UINT>(barriers.size()), barriers.data());
const D3D12_TEXTURE_COPY_LOCATION sourceLocation{
.pResource = source.resource.Get(),
.Type = D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX,
.SubresourceIndex = 0,
};
const D3D12_TEXTURE_COPY_LOCATION destinationLocation{
.pResource = destination.resource.Get(),
.Type = D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX,
.SubresourceIndex = 0,
};
const D3D12_BOX sourceBox{0, 0, 0, source.width, source.height, 1};
list->CopyTextureRegion(&destinationLocation, 0, 0, 0, &sourceLocation, &sourceBox);
const std::array restore{
D3D12_RESOURCE_BARRIER{
.Type = D3D12_RESOURCE_BARRIER_TYPE_TRANSITION,
.Flags = D3D12_RESOURCE_BARRIER_FLAG_NONE,
.Transition = {source.resource.Get(), D3D12_RESOURCE_BARRIER_ALL_SUBRESOURCES,
D3D12_RESOURCE_STATE_COPY_SOURCE, D3D12_RESOURCE_STATE_COMMON},
},
D3D12_RESOURCE_BARRIER{
.Type = D3D12_RESOURCE_BARRIER_TYPE_TRANSITION,
.Flags = D3D12_RESOURCE_BARRIER_FLAG_NONE,
.Transition = {destination.resource.Get(), D3D12_RESOURCE_BARRIER_ALL_SUBRESOURCES,
D3D12_RESOURCE_STATE_COPY_DEST,
D3D12_RESOURCE_STATE_RENDER_TARGET},
},
};
list->ResourceBarrier(static_cast<UINT>(restore.size()), restore.data());
}
if (FAILED(list->Close())) {
Log.error("Could not close the D3D12 stereo copy command list");
return false;
}
ID3D12CommandList* lists[]{list.Get()};
m_queue->ExecuteCommandLists(1, lists);
m_gpuIdle = false;
const uint64_t fenceValue = ++m_nextFenceValue;
const HRESULT signalResult = m_queue->Signal(m_fence.Get(), fenceValue);
m_commands.push_back({FAILED(signalResult) ? kUnfencedSubmission : fenceValue,
std::move(allocator), std::move(list),
std::move(sources), std::move(destinations)});
if (FAILED(signalResult)) {
// ExecuteCommandLists has already transferred work to the queue. Keep
// every command/resource reference alive even though there is no usable
// completion value; shutdown will retry with a queue-tail fence and leak
// this small bridge on an unrecoverable device/queue failure.
Log.error("Could not signal the D3D12 stereo copy fence");
return false;
}
m_lastExternalFenceValue = fenceValue;
return true;
}
void CollectCompletedCommandsLocked() noexcept {
const uint64_t completed = m_fence ? m_fence->GetCompletedValue() : 0;
std::erase_if(m_commands, [completed](const InFlightCommand& command) {
return command.fenceValue != kUnfencedSubmission && command.fenceValue <= completed;
});
}
void ClearFrameLocked() noexcept {
for (auto& target : m_targets) {
target = {};
}
m_frameToken = 0;
m_targetCount = 0;
m_framePending = false;
m_encoded = false;
}
void PublishAndClearFrameLocked(uint64_t token, bool success) noexcept {
// Publication is part of the bridge state transition: once another thread
// can observe that this token is no longer cancellable, its submission
// result must already be visible. The OpenXR callback only takes the
// backend submission mutex; no backend path holds that mutex while entering
// this bridge, so keeping m_mutex here preserves the lock order.
Notify(token, success);
ClearFrameLocked();
}
void Notify(uint64_t token, bool success) noexcept {
if (m_callback != nullptr) {
m_callback(token, success, m_userdata);
}
}
bool WaitForGpuLocked() noexcept {
if (m_gpuIdle) {
return true;
}
if (!m_queue || !m_fence) {
return m_commands.empty();
}
const uint64_t value = ++m_nextFenceValue;
if (FAILED(m_queue->Signal(m_fence.Get(), value))) {
Log.error("Could not signal a D3D12 queue-tail fence during stereo bridge shutdown");
return false;
}
if (m_fence->GetCompletedValue() >= value) {
m_commands.clear();
m_gpuIdle = true;
return true;
}
HANDLE event = CreateEventW(nullptr, FALSE, FALSE, nullptr);
if (event == nullptr) {
Log.error("Could not create the D3D12 stereo shutdown fence event");
return false;
}
bool complete = false;
if (SUCCEEDED(m_fence->SetEventOnCompletion(value, event))) {
complete = WaitForSingleObject(event, 5000) == WAIT_OBJECT_0;
}
CloseHandle(event);
if (!complete) {
Log.error("Timed out waiting for the D3D12 stereo queue to become idle");
return false;
}
m_commands.clear();
m_gpuIdle = true;
return true;
}
std::mutex m_mutex;
ComPtr<ID3D12Device> m_device;
ComPtr<ID3D12CommandQueue> m_queue;
ComPtr<ID3D12Fence> m_fence;
wgpu::SharedFence m_webgpuFence;
std::array<IntermediateEye, AURORA_D3D12_STEREO_MAX_TARGETS> m_intermediates{};
std::array<PendingTarget, AURORA_D3D12_STEREO_MAX_TARGETS> m_targets{};
std::vector<InFlightCommand> m_commands;
AuroraD3D12StereoSubmittedCallback m_callback = nullptr;
void* m_userdata = nullptr;
uint64_t m_frameToken = 0;
uint64_t m_nextFenceValue = 0;
uint64_t m_lastExternalFenceValue = 0;
uint32_t m_targetCount = 0;
bool m_framePending = false;
bool m_encoded = false;
bool m_gpuIdle = true;
};
std::unique_ptr<StereoBridge> g_bridge;
bool sink_encode(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame,
void* userdata) noexcept {
return static_cast<StereoBridge*>(userdata)->Encode(encoder, frame);
}
void sink_submitted(const stereo::SinkFrame& frame, void* userdata) noexcept {
static_cast<StereoBridge*>(userdata)->Submitted(frame);
}
} // namespace
} // namespace aurora::d3d12_interop
bool aurora_d3d12_get_native_handles(AuroraD3D12NativeHandles* handles) {
if (handles == nullptr) {
return false;
}
*handles = {};
aurora::d3d12_interop::NativeObjects objects;
if (!aurora::d3d12_interop::get_native_objects(objects)) {
return false;
}
const int64_t colorFormat =
aurora::d3d12_interop::to_dxgi_format(aurora::webgpu::g_graphicsConfig.surfaceConfiguration.format);
if (colorFormat == DXGI_FORMAT_UNKNOWN) {
return false;
}
const LUID luid = objects.device->GetAdapterLuid();
*handles = {
.device = objects.device.Get(),
.queue = objects.queue.Get(),
.colorDxgiFormat = colorFormat,
.adapterLuidLow = luid.LowPart,
.adapterLuidHigh = luid.HighPart,
};
return true;
}
bool aurora_d3d12_enable_stereo_bridge(AuroraD3D12StereoSubmittedCallback submitted,
void* userdata) {
using namespace aurora::d3d12_interop;
if (g_bridge || submitted == nullptr) {
return false;
}
NativeObjects objects;
if (!get_native_objects(objects)) {
return false;
}
auto bridge = std::make_unique<StereoBridge>(std::move(objects), submitted, userdata);
if (!bridge->Initialize()) {
return false;
}
aurora::stereo::set_sink(sink_encode, sink_submitted, bridge.get());
g_bridge = std::move(bridge);
return true;
}
bool aurora_d3d12_set_stereo_targets(uint64_t frameToken,
const AuroraD3D12StereoTarget* targets,
uint32_t targetCount) {
using namespace aurora::d3d12_interop;
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount);
}
bool aurora_d3d12_cancel_stereo_targets(uint64_t frameToken) {
using namespace aurora::d3d12_interop;
return g_bridge && g_bridge->CancelBeforeEncode(frameToken);
}
bool aurora_d3d12_disable_stereo_bridge() {
using namespace aurora::d3d12_interop;
if (!g_bridge) {
return true;
}
aurora::stereo::set_sink(nullptr, nullptr, nullptr);
g_bridge->CancelPending();
if (!g_bridge->PrepareForDestruction()) {
// An already-enqueued command has no trustworthy completion marker. Keep
// the bridge, queue, command lists and resource references alive for the
// rest of the process rather than freeing memory the GPU may still touch.
(void)g_bridge.release();
return false;
}
g_bridge.reset();
return true;
}
#else
bool aurora_d3d12_get_native_handles(AuroraD3D12NativeHandles* handles) {
if (handles != nullptr) {
*handles = {};
}
return false;
}
bool aurora_d3d12_enable_stereo_bridge(AuroraD3D12StereoSubmittedCallback, void*) {
return false;
}
bool aurora_d3d12_set_stereo_targets(uint64_t, const AuroraD3D12StereoTarget*, uint32_t) {
return false;
}
bool aurora_d3d12_cancel_stereo_targets(uint64_t) { return false; }
bool aurora_d3d12_disable_stereo_bridge() { return true; }
#endif
+18 -1
View File
@@ -540,11 +540,27 @@ bool initialize(AuroraBackend auroraBackend) {
.requiredFeatureCount = requiredInstanceFeatures.size(), .requiredFeatureCount = requiredInstanceFeatures.size(),
.requiredFeatures = requiredInstanceFeatures.data(), .requiredFeatures = requiredInstanceFeatures.data(),
}; };
#ifdef WEBGPU_DAWN
// Dawn hides its D3D12 shared-resource feature from adapter enumeration
// unless unsafe APIs are exposed by the instance. OpenXR's same-device
// bridge is the only Aurora path that needs it, so keep the wider Dawn API
// surface scoped to an explicitly requested XR interop instance.
const std::array xrInteropInstanceToggles{"allow_unsafe_apis"};
wgpu::DawnTogglesDescriptor instanceToggles({
.enabledToggleCount = xrInteropInstanceToggles.size(),
.enabledToggles = xrInteropInstanceToggles.data(),
});
if (g_config.xrInterop) {
instanceDescriptor.nextInChain = &instanceToggles;
Log.info("Enabling Dawn unsafe APIs for OpenXR D3D12 resource interop");
}
#endif
#if defined(WEBGPU_DAWN) && !defined(__MINGW32__) #if defined(WEBGPU_DAWN) && !defined(__MINGW32__)
// DawnNative.h's C++ constructor has an MSVC ABI that cannot cross into llvm-mingw, and the // DawnNative.h's C++ constructor has an MSVC ABI that cannot cross into llvm-mingw, and the
// descriptor only restates Dawn's defaults, so use the public WebGPU descriptor here. // descriptor only restates Dawn's defaults, so use the public WebGPU descriptor here.
dawn::native::DawnInstanceDescriptor dawnInstanceDescriptor; dawn::native::DawnInstanceDescriptor dawnInstanceDescriptor;
dawnInstanceDescriptor.backendValidationLevel = dawn::native::BackendValidationLevel::Disabled; dawnInstanceDescriptor.backendValidationLevel = dawn::native::BackendValidationLevel::Disabled;
dawnInstanceDescriptor.nextInChain = instanceDescriptor.nextInChain;
instanceDescriptor.nextInChain = &dawnInstanceDescriptor; instanceDescriptor.nextInChain = &dawnInstanceDescriptor;
#endif #endif
g_instance = wgpu::CreateInstance(&instanceDescriptor); g_instance = wgpu::CreateInstance(&instanceDescriptor);
@@ -680,7 +696,8 @@ bool initialize(AuroraBackend auroraBackend) {
} }
#if defined(WEBGPU_DAWN) && defined(_WIN32) #if defined(WEBGPU_DAWN) && defined(_WIN32)
if (g_config.xrInterop && g_backendType == wgpu::BackendType::D3D12 && if (g_config.xrInterop && g_backendType == wgpu::BackendType::D3D12 &&
feature == wgpu::FeatureName::SharedTextureMemoryD3D12Resource) { (feature == wgpu::FeatureName::SharedTextureMemoryD3D12Resource ||
feature == wgpu::FeatureName::SharedFenceDXGISharedHandle)) {
requiredFeatures.push_back(feature); requiredFeatures.push_back(feature);
} }
#endif #endif
+1
View File
@@ -17,6 +17,7 @@ if (AURORA_ENABLE_GX)
add_executable(gx_fifo_tests add_executable(gx_fifo_tests
gx_fifo_test.cpp gx_fifo_test.cpp
gx_test_stubs.cpp gx_test_stubs.cpp
stereo_replay_test.cpp
texture_bind_group_cache_key_test.cpp texture_bind_group_cache_key_test.cpp
../lib/gfx/efb_ram_encoder.cpp ../lib/gfx/efb_ram_encoder.cpp
# GX API implementations (encoders) # GX API implementations (encoders)
+8
View File
@@ -49,6 +49,7 @@ std::recursive_mutex& renderer_gpu_mutex() noexcept {
} // namespace aurora } // namespace aurora
extern "C" bool aurora_wait_for_frame_worker_for(uint32_t) { return true; } extern "C" bool aurora_wait_for_frame_worker_for(uint32_t) { return true; }
extern "C" void aurora_quiesce_frame_worker() {}
// --- aurora::log_internal --- // --- aurora::log_internal ---
namespace aurora { namespace aurora {
@@ -453,6 +454,13 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
record.persistentCopy = persistentCopy; record.persistentCopy = persistentCopy;
testing::s_resolvePassRecords.push_back(std::move(record)); testing::s_resolvePassRecords.push_back(std::move(record));
} }
// The recorded resolve is the display copy's; the flag only steers render-pass
// bookkeeping that lives in common.cpp, which this target does not compile.
void mark_last_resolve_as_display_copy() noexcept {}
void set_stereo_stop_at_display_copy(bool value) noexcept {}
bool get_stereo_stop_at_display_copy() noexcept { return true; }
void set_stereo_skip_copy_clears(bool value) noexcept {}
bool get_stereo_skip_copy_clears() noexcept { return true; }
void queue_palette_conv(tex_palette_conv::ConvRequest req) {} void queue_palette_conv(tex_palette_conv::ConvRequest req) {}
void begin_offscreen(uint32_t width, uint32_t height) {} void begin_offscreen(uint32_t width, uint32_t height) {}
void end_offscreen() {} void end_offscreen() {}
+41
View File
@@ -0,0 +1,41 @@
#include "gfx/stereo_replay.hpp"
#include <gtest/gtest.h>
namespace aurora::gfx::stereo_replay {
namespace {
TEST(StereoReplayTest, EyeFrustumPreservesGameDepthMapping) {
const Mat4x4<float> game{
{10.0f, 11.0f, 12.0f, 13.0f},
{20.0f, 21.0f, 22.0f, 23.0f},
{30.0f, 31.0f, 32.0f, 33.0f},
{40.0f, 41.0f, 42.0f, 43.0f},
};
const Mat4x4<float> eye{
{1.1f, 1.2f, 1.3f, 1.4f},
{2.1f, 2.2f, 2.3f, 2.4f},
{3.1f, 3.2f, 3.3f, 3.4f},
{4.1f, 4.2f, 4.3f, 4.4f},
};
const auto result = compose_projection(eye, game);
EXPECT_FLOAT_EQ(result.m0[0], eye.m0[0]);
EXPECT_FLOAT_EQ(result.m0[2], eye.m0[2]);
EXPECT_FLOAT_EQ(result.m1[1], eye.m1[1]);
EXPECT_FLOAT_EQ(result.m1[2], eye.m1[2]);
for (size_t row = 0; row < 4; ++row) {
for (size_t column = 0; column < 4; ++column) {
const bool frustumTerm =
(row == 0 && (column == 0 || column == 2)) ||
(row == 1 && (column == 1 || column == 2));
if (!frustumTerm) {
EXPECT_FLOAT_EQ(result[row][column], game[row][column]);
}
}
}
}
} // namespace
} // namespace aurora::gfx::stereo_replay