Merge remote-tracking branch 'upstream/openxr-work' into codex/standalone-selective-integration

# Conflicts:
#	docs/quest-port.md
This commit is contained in:
darthcircuit committed 2026-09-23 11:28:20 -06:00
commit 0e52cfb726
141 files changed
+14235 -1090

No files matched your search

+1 -1
View File
@@ -38,7 +38,7 @@ if (AURORA_ENABLE_GX)
# OpenXR-enabled runtime can fail back to desktop mode at runtime instead
# of producing unresolved interop references.
if (CMAKE_SYSTEM_NAME STREQUAL Windows)
target_sources(aurora_core PRIVATE lib/webgpu/d3d12_interop.cpp)
target_sources(aurora_core PRIVATE lib/webgpu/d3d12_interop.cpp lib/webgpu/vulkan_win32_interop.cpp)
endif ()
# Android/Vulkan counterpart: the AHardwareBuffer stereo bridge. The file
# compiles to C ABI stubs on every other platform so the runtime's OpenXR
+79
View File
@@ -116,6 +116,54 @@ typedef enum {
// accepted through aurora_end_frame_tagged() with an exact matching tag.
#define AURORA_STEREO_CONTENT_TAG_UNKNOWN UINT64_MAX
/**
* VR cockpit overlay: tracked hands and, when the vehicle's own wheel cannot be
* animated, a synthetic steering wheel or handlebar. Everything is in metres
* in a seated frame (+X right, +Y up, -Z forward) whose origin is the headset's
* immersive base position. Aurora draws it per eye after the scene, depth-tested
* against the scene with the scene's own depth mapping.
*/
typedef struct {
bool tracked;
bool held;
float squeeze;
float seatFromGrip[12];
} AuroraCockpitHand;
typedef struct {
bool active;
float wheelAngle;
// The vehicle's own wheel is animated in the scene, so no synthetic wheel is drawn.
bool nativeWheel;
bool bike;
float handlebarRadius;
// World units per metre used to build this packet's eye transforms.
float unitsPerMeter;
float seatFromHandlebar[12];
float eyeFromSeat[AURORA_STEREO_EYE_COUNT][12];
AuroraCockpitHand hands[2];
} AuroraCockpit;
typedef struct {
float position[3];
int16_t joints[4];
float weights[4];
} AuroraVRHandVertex;
/**
* Shows the headset settings panel (aurora_imgui_set_stereo_overlay) as the
* OpenXR backend's own compositor quad layer instead of drawing it into the
* eyes, so the eye resolution no longer limits its text. The backend then asks
* for the panel image as an extra stereo target. Any thread.
*/
void aurora_set_stereo_panel_layer(bool enabled);
// Copies optional runtime-provided hand meshes (XR_FB_hand_tracking_mesh, 26
// joints). Null clears to the procedural glove. Bind poses: x,y,z,w,px,py,pz.
void aurora_set_vr_hand_mesh(uint32_t hand, const AuroraVRHandVertex* vertices, uint32_t vertexCount,
const uint16_t* indices, uint32_t indexCount, const float* bindPoses,
const int32_t* parents, uint32_t jointCount);
/**
* Stereo data for one sealed GX frame. frameToken is opaque to Aurora and is
* forwarded unchanged to the internal stereo output sink. contentTag must
@@ -130,6 +178,8 @@ typedef struct {
// Predicted display time converted to std::chrono::steady_clock nanoseconds.
// Zero disables temporal interpolation for this packet.
uint64_t displayTimeNanos;
// Optional; inactive when zero-initialised.
AuroraCockpit cockpit;
} AuroraStereoFrame;
/**
@@ -218,6 +268,17 @@ void aurora_end_frame();
// Seal the current frame with an opaque application safety tag. Aurora rejects
// an immersive provider packet unless its contentTag matches this exact frame.
void aurora_end_frame_tagged(uint64_t contentTag);
// aurora_end_frame_tagged() plus the host-owned ImGui frame to present with it (the handle from
// aurora_imgui_host_frame_end(), which this call consumes; NULL presents no host ImGui frame).
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame);
// When the host pumps SDL events itself (aurora_update() on the window's thread) and drives
// begin/end frame from another thread, this stops those calls from pumping events.
void aurora_set_host_event_pump(bool hostPumps);
typedef void (*AuroraFrameLogCallback)(char* buffer, uint32_t bufferSize, double windowSeconds,
uint32_t frames);
// Called with each five-second frame-rate log window (where that log is enabled); a non-empty
// buffer is logged as one extra line.
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback);
/**
* Relocates the immersive camera for the frame about to be sealed.
*
@@ -236,10 +297,28 @@ void aurora_end_frame_tagged(uint64_t contentTag);
* provider, which cannot know which frame will consume its packet.
*/
void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]);
// As above, also naming the world units per metre the anchor was built with.
// The sealed frame then owns that scale: each eye's head/IPD translation is
// rescaled from the packet's AuroraCockpit::unitsPerMeter to it.
void aurora_set_stereo_scene_anchor_scaled(const float anchorFromScene[12], float unitsPerMeter);
// Select Player 1's subview for immersive replay of 2-4 local screens.
// Producer-thread, per-frame metadata, consumed by the next end_frame call.
// One (the default) keeps full-frame replay. Desktop rendering is unaffected.
void aurora_set_stereo_local_player_count(uint32_t count);
/**
* VR native steering wheel: replacement position arrays for the local vehicle.
*
* GX (command processor) thread only, in order with the frame's draws: a host
* that runs GX on its own thread must post these there. `source` is the host
* pointer the game binds with GXSetArray; `replacement` is copied (at most 64
* KiB) and applies solely to draws that bind that array with a position matrix
* equal to `modelView` (row-major 3x4), so shared opponent models stay intact.
* Clearing reports how many draws the previous set matched.
*/
void aurora_clear_native_wheel_vertices(void);
void aurora_set_native_wheel_vertices(const void* source, const void* replacement, uint32_t size,
const float* modelView);
uint32_t aurora_native_wheel_draw_count(void);
typedef void (*AuroraFrameWorkerWaitCallback)();
// Called from the producer thread at bounded intervals while Aurora waits for
// the asynchronous frame worker. The callback must not enter Aurora.
@@ -64,6 +64,17 @@ bool aurora_d3d12_set_stereo_targets(uint64_t frameToken,
const AuroraD3D12StereoTarget* targets,
uint32_t targetCount);
/**
* The same, plus the headset settings panel's quad-layer image when `panel` is
* not null (aurora_set_stereo_panel_layer): Aurora copies the panel into it, or
* a transparent image while the panel is not showing, with the eyes and under
* the same completion callback.
*/
bool aurora_d3d12_set_stereo_targets_with_panel(uint64_t frameToken,
const AuroraD3D12StereoTarget* targets,
uint32_t targetCount,
const AuroraD3D12StereoTarget* panel);
/**
* Withdraws frameToken only while its target has not been encoded. This is
* safe to race with Aurora's frame worker: false means the worker already owns
@@ -0,0 +1,33 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#pragma once
#include <stdint.h>
// Versioned C ABI: no STL objects or compiler-specific C++ symbols cross the
// MSVC Dawn DLL / LLVM-MinGW runtime boundary. Vulkan handles are borrowed.
#define AURORA_DAWN_VULKAN_ABI 1
#ifdef __cplusplus
extern "C" {
#endif
typedef struct {
void* userdata;
int32_t (*createInstance)(void*, void* getProc, const void* info, const void* allocator, void** instance);
int32_t (*createDevice)(void*, void* getProc, void* physical, const void* info, const void* allocator, void** device);
int32_t (*getPhysicalDevice)(void*, void* instance, void** physical);
} AuroraDawnVulkanHooks;
typedef struct {
void* instance;
void* physicalDevice;
void* device;
uint32_t queueFamily;
uint32_t queueIndex;
} AuroraDawnVulkanHandles;
typedef uint32_t (*AuroraDawnVulkanVersionFn)(void);
typedef int (*AuroraDawnVulkanConfigureFn)(const AuroraDawnVulkanHooks*);
typedef int (*AuroraDawnVulkanHandlesFn)(void* device, AuroraDawnVulkanHandles*);
typedef void* (*AuroraDawnVulkanWrapFn)(void* device, const void* textureDescriptor, uint64_t image);
typedef int (*AuroraDawnVulkanReleaseFn)(void* device, void* const* textures, uint32_t count);
typedef void* (*AuroraDawnVulkanLockFn)(void* device);
typedef void (*AuroraDawnVulkanUnlockFn)(void* guard);
typedef int (*AuroraDawnVulkanDrainFn)(void* device);
#ifdef __cplusplus
}
#endif
+8
View File
@@ -25,6 +25,14 @@ ImTextureID aurora_imgui_add_texture(uint32_t width, uint32_t height, const void
// producer before the frame is sealed, and leave the draw data untouched until the frame worker is
// done with that frame (aurora_wait_for_frame_worker). Null hides the panel.
void aurora_imgui_set_stereo_overlay(ImDrawData* drawData, float widthFraction);
// Host-owned ImGui frames for the desktop overlay. Begin starts the next frame on the calling
// thread (which must be the window's thread, since the SDL backend reads the window there); end
// renders it and returns a handle to a private copy of its draw data, which aurora_end_frame_ex()
// consumes. A handle that is never presented is freed with aurora_imgui_host_frame_release().
// From the first begin on, aurora no longer starts ImGui frames itself.
void aurora_imgui_host_frame_begin(void);
void* aurora_imgui_host_frame_end(void);
void aurora_imgui_host_frame_release(void* imguiFrame);
#ifdef __cplusplus
}
@@ -0,0 +1,32 @@
// SPDX-License-Identifier: GPL-3.0-or-later
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
#pragma once
#include <cstdint>
#include <cstring>
#include <span>
namespace aurora {
// Every referenced position that changes must belong to the local body matrix.
// Unchanged positions may use other joints in the same indexed draw (wing/engine).
inline bool NativeWheelDrawMatches(std::span<const uint8_t> original,
std::span<const uint8_t> replacement, uint32_t positionStride,
std::span<const uint8_t> vertices, uint32_t vertexStride, uint32_t positionOffset,
uint32_t indexBytes, uint16_t matchingMatrices) {
if(!positionStride || !vertexStride || (indexBytes!=1 && indexBytes!=2) ||
positionOffset>=vertexStride || indexBytes>vertexStride-positionOffset ||
original.size()!=replacement.size() || vertices.size()%vertexStride || !matchingMatrices) return false;
bool changed=false;
for(size_t start=0;start<vertices.size();start+=vertexStride) {
const auto* vertex=vertices.data()+start;
const uint32_t index=indexBytes==1 ? vertex[positionOffset]
: (uint32_t(vertex[positionOffset])<<8)|vertex[positionOffset+1];
const size_t offset=size_t(index)*positionStride;
if(offset>original.size() || positionStride>original.size()-offset) return false;
if(std::memcmp(original.data()+offset,replacement.data()+offset,positionStride)==0) continue;
const uint32_t matrix=vertex[0]/3u;
if(vertex[0]%3u || matrix>=16 || !(matchingMatrices&(1u<<matrix))) return false;
changed=true;
}
return changed;
}
}
@@ -27,6 +27,8 @@ extern "C" {
*/
enum { AURORA_VULKAN_STEREO_MAX_TARGETS = 2 };
// The eyes, then the settings panel's quad-layer image when one was given.
enum { AURORA_VULKAN_STEREO_MAX_RELEASES = AURORA_VULKAN_STEREO_MAX_TARGETS + 1 };
/**
* Borrowed facts about Aurora's Dawn Vulkan device. colorVkFormat is the
@@ -109,6 +111,18 @@ bool aurora_vulkan_set_stereo_targets(uint64_t frameToken,
const AuroraVulkanStereoTarget* targets,
uint32_t targetCount);
/**
* The same, plus the headset settings panel's quad-layer buffer when `panel` is
* not null (aurora_set_stereo_panel_layer): Aurora copies the panel into it, or
* a transparent image while the panel is not showing, with the eyes. Its
* release entry follows the eyes' in the submitted callback, whose
* releaseCount then counts it too.
*/
bool aurora_vulkan_set_stereo_targets_with_panel(uint64_t frameToken,
const AuroraVulkanStereoTarget* targets,
uint32_t targetCount,
const AuroraVulkanStereoTarget* panel);
/**
* Withdraws frameToken only while its targets have not been encoded. Semantics
* match aurora_d3d12_cancel_stereo_targets: false means the worker already
@@ -0,0 +1,24 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#pragma once
#include <aurora/dawn_vulkan_abi.h>
#include <aurora/d3d12_interop.h>
#ifdef __cplusplus
extern "C" {
#endif
// Same pending-target/callback contract as D3D12. resource carries a VkImage
// encoded as a pointer-sized value; colorDxgiFormat carries a VkFormat.
bool aurora_vulkan_win32_configure(const AuroraDawnVulkanHooks* hooks);
bool aurora_vulkan_win32_get_handles(AuroraDawnVulkanHandles* handles, int64_t* colorFormat);
bool aurora_vulkan_win32_enable(AuroraD3D12StereoSubmittedCallback submitted, void* userdata);
bool aurora_vulkan_win32_set_targets(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count);
// Plus the headset settings panel's quad-layer image when panel is not null, as
// aurora_d3d12_set_stereo_targets_with_panel.
bool aurora_vulkan_win32_set_targets_with_panel(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count,
const AuroraD3D12StereoTarget* panel);
bool aurora_vulkan_win32_cancel(uint64_t token);
bool aurora_vulkan_win32_disable();
void* aurora_vulkan_win32_lock_queue();
void aurora_vulkan_win32_unlock_queue(void* guard);
#ifdef __cplusplus
}
#endif
+184 -30
View File
@@ -28,6 +28,7 @@
#include "gfx/pipeline_cache.hpp"
#endif
#if defined(__ANDROID__)
#include <pthread.h>
#include <unistd.h>
#endif
#include "system_info.hpp"
@@ -69,6 +70,9 @@ AuroraConfig g_config;
uint32_t g_sdlCustomEventsStart;
char g_gameName[4];
std::atomic<AuroraFrameWorkerWaitCallback> g_frameWorkerWaitCallback{nullptr};
// aurora_set_host_event_pump(): the host pumps SDL itself, from the window's thread.
std::atomic_bool g_hostEventPump{false};
std::atomic<AuroraFrameLogCallback> g_frameLogCallback{nullptr};
// Presentation schedule for the frame being sealed, set by the producer. Jobs carry absolute
// deadlines derived from it, so the presenter cannot drift. Zero means present when ready.
std::atomic<uint64_t> g_presentScheduleBaseNanos{0};
@@ -109,6 +113,9 @@ struct StereoSceneAnchor {
};
bool active = false;
uint32_t localPlayerCount = 1;
// World units per metre the anchor was built with, or zero when the packet's
// own scale applies (aurora_set_stereo_scene_anchor_scaled).
float unitsPerMeter = 0.f;
};
// Producer thread only, between aurora_set_stereo_scene_anchor() and the seal
// that consumes it. Cleared at every seal so a producer that stops publishing
@@ -266,7 +273,8 @@ enum class ImGuiFramePolicy {
bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy = ImGuiFramePolicy::Immediate,
bool* imguiNewFrameOwed = nullptr) noexcept;
bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewFrameOwed) noexcept;
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor) noexcept;
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor,
imgui::HostFramePtr hostImGuiFrame) noexcept;
// The two publication points of a frame-worker cycle, cleared together under `mutex`. Sealed:
// producer-shared renderer state is free again. Done: slots encoded, presented, ImGui restarted.
@@ -288,6 +296,7 @@ struct FrameWorkerState {
// belong to that exact queued frame, not to the producer's next frame.
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
StereoSceneAnchor sceneAnchor{};
imgui::HostFramePtr hostImGuiFrame;
// Readiness is polled thousands of times per frame, so these flags double as a publication
// barrier. `sealed` is released before `ready`, and both are cleared under `mutex`.
std::atomic_bool sealed{true};
@@ -338,7 +347,7 @@ bool frame_worker_requested() noexcept {
#ifdef AURORA_ENABLE_GX
// Returns false when a stop request was observed mid-cycle.
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
const StereoSceneAnchor& sceneAnchor) noexcept;
void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept;
#endif
@@ -350,6 +359,9 @@ void frame_worker_main() noexcept {
}
#if defined(__ANDROID__)
g_frameWorkerNativeThreadId.store(static_cast<uint32_t>(gettid()), std::memory_order_release);
// A thread inherits its creator's name, and the producer that starts this
// worker may itself be a named thread; profiles should tell the two apart.
pthread_setname_np(pthread_self(), "aurora worker");
#endif
#ifdef AURORA_ENABLE_GX
@@ -361,6 +373,7 @@ void frame_worker_main() noexcept {
for (;;) {
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
StereoSceneAnchor sceneAnchor{};
imgui::HostFramePtr hostImGuiFrame;
bool stereoOnly = false;
{
std::unique_lock lock(g_frameWorker.mutex);
@@ -377,6 +390,7 @@ void frame_worker_main() noexcept {
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
sceneAnchor = g_frameWorker.sceneAnchor;
g_frameWorker.sceneAnchor = {};
hostImGuiFrame = std::move(g_frameWorker.hostImGuiFrame);
g_frameWorker.jobPending = false;
}
@@ -396,7 +410,7 @@ void frame_worker_main() noexcept {
g_frameWorker.cv.notify_all();
continue;
}
if (!run_frame_worker_cycle(sealedFrame, contentTag, sceneAnchor)) {
if (!run_frame_worker_cycle(sealedFrame, contentTag, std::move(hostImGuiFrame), sceneAnchor)) {
break;
}
#else
@@ -619,7 +633,7 @@ void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height
const uint32_t samples = webgpu::g_graphicsConfig.msaaSamples;
if (target.color.texture && target.requestedWidth == width && target.requestedHeight == height &&
target.samples == samples && target.colorFormat == webgpu::g_graphicsConfig.surfaceConfiguration.format &&
target.depthFormat == webgpu::g_graphicsConfig.depthFormat) {
target.depthFormat == wgpu::TextureFormat::Depth24PlusStencil8) {
return;
}
@@ -629,19 +643,21 @@ void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height
target.resolvedColor = webgpu::create_render_texture(target.color.size.width, target.color.size.height, false);
}
// The cockpit uses one stencil bit to survive later depth-disabled HUD draws.
// Keep this attachment eye-only; native EFB depth sampling is unchanged.
const wgpu::TextureDescriptor depthDescriptor{
.label = eyeIndex == 0 ? "Stereo left eye depth" : "Stereo right eye depth",
.usage = wgpu::TextureUsage::RenderAttachment,
.dimension = wgpu::TextureDimension::e2D,
.size = target.color.size,
.format = webgpu::g_graphicsConfig.depthFormat,
.format = wgpu::TextureFormat::Depth24PlusStencil8,
.mipLevelCount = 1,
.sampleCount = samples,
};
target.depth.texture = g_device.CreateTexture(&depthDescriptor);
target.depth.view = target.depth.texture.CreateView();
target.depth.size = target.color.size;
target.depth.format = webgpu::g_graphicsConfig.depthFormat;
target.depth.format = wgpu::TextureFormat::Depth24PlusStencil8;
target.requestedWidth = width;
target.requestedHeight = height;
target.samples = samples;
@@ -702,6 +718,27 @@ std::optional<AuroraStereoFrame> request_stereo_frame(uint32_t logicalFrame, uin
return std::nullopt;
}
}
// The cockpit overlay is optional: a bad one is dropped, never the frame.
if (frame.cockpit.active) {
const auto& cockpit = frame.cockpit;
bool valid = finite(&cockpit.wheelAngle, 1) && finite(&cockpit.handlebarRadius, 1) &&
finite(&cockpit.unitsPerMeter, 1) && cockpit.unitsPerMeter > 0.f &&
finite(cockpit.seatFromHandlebar, 12);
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
valid = valid && finite(cockpit.eyeFromSeat[eye], 12);
}
for (const auto& hand : cockpit.hands) {
valid = valid && finite(&hand.squeeze, 1) && finite(hand.seatFromGrip, 12);
}
if (!valid) {
static bool cockpitRejectionLogged = false;
if (!cockpitRejectionLogged) {
cockpitRejectionLogged = true;
Log.warn("Stereo frame {} carries a non-finite VR cockpit; drawing it without the cockpit", logicalFrame);
}
frame.cockpit = {};
}
}
return frame;
}
@@ -709,6 +746,16 @@ gfx::StereoReplayFrame make_stereo_replay_frame(const AuroraStereoFrame& input,
Mat3x4<float> anchorFromScene;
std::memcpy(&anchorFromScene, sceneAnchor.anchorFromScene.data(), sizeof(anchorFromScene));
gfx::StereoReplayFrame replay{};
replay.cockpit = input.cockpit;
// The sealed guest frame owns its scale. The packet may have been sampled
// just before a change of scale (a character swap, a lightning strike), so
// only its head/IPD translation is rescaled to the frame's.
const float frameUnits = sceneAnchor.active && sceneAnchor.unitsPerMeter > 0.f ? sceneAnchor.unitsPerMeter
: input.cockpit.unitsPerMeter;
const float unitRatio = input.cockpit.unitsPerMeter > 0.f && frameUnits > 0.f
? frameUnits / input.cockpit.unitsPerMeter
: 1.f;
replay.cockpit.unitsPerMeter = frameUnits;
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
ensure_stereo_eye_target(eye, input.eyes[eye].width, input.eyes[eye].height);
const auto& owned = g_stereoEyeTargets[eye];
@@ -723,9 +770,15 @@ gfx::StereoReplayFrame make_stereo_replay_frame(const AuroraStereoFrame& input,
.copySourceDepthView = owned.depth.view,
.size = owned.color.size,
.msaaSamples = webgpu::g_graphicsConfig.msaaSamples,
.depthFormat = owned.depth.format,
};
std::memcpy(&view.projection, input.eyes[eye].projection, sizeof(view.projection));
std::memcpy(&view.viewFromCenter, input.eyes[eye].viewFromCenter, sizeof(view.viewFromCenter));
if (unitRatio != 1.f) {
view.viewFromCenter.m0[3] *= unitRatio;
view.viewFromCenter.m1[3] *= unitRatio;
view.viewFromCenter.m2[3] *= unitRatio;
}
// World draws already carry the recorded camera, so they need the anchor
// folded in; the virtual screen is authored in the anchored camera's space
// and keeps viewFromCenter.
@@ -756,6 +809,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen),
};
{
const auto pass = encoder.BeginRenderPass(&descriptor);
@@ -1274,6 +1328,7 @@ bool present_presentation_job(const PresentationJob& job) {
.label = "Presentation copy pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(webgpu::g_CopyPipeline);
@@ -1554,7 +1609,7 @@ void publish_stereo_screen_aspects(const webgpu::PresentSource& presentSource, c
// gfx::begin_frame() may already have cleared the display-copy override.
void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const webgpu::PresentSource& presentSource,
const PresentationImage& image, bool includeImGui,
MirrorPlan plan = MirrorPlan::Mono) {
MirrorPlan plan = MirrorPlan::Mono, const ImDrawData* hostImGuiData = nullptr) {
ZoneScoped;
auto viewport = webgpu::calculate_present_viewport(image.texture.size.width, image.texture.size.height,
presentSource.size.width, presentSource.size.height);
@@ -1576,6 +1631,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Interpolation snapshot pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
const auto imageWidth = static_cast<float>(image.texture.size.width);
@@ -1627,11 +1683,16 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
.label = "Snapshot ImGui pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
};
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
static_cast<float>(image.texture.size.height), 0.f, 1.f);
imgui::render(pass);
if (hostImGuiData != nullptr) {
imgui::render(pass, hostImGuiData);
} else {
imgui::render(pass);
}
pass.End();
}
}
@@ -1674,7 +1735,7 @@ bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy, bool* imgui
ZoneScoped;
#ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost();
if (pumpEvents) {
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
window::pump_events();
}
const bool surfaceReconfigurePending = g_surfaceReconfigurePending.load(std::memory_order_acquire);
@@ -1731,7 +1792,9 @@ bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewF
std::lock_guard gpuLock(g_rendererGpuMutex);
// Note the debt before gfx::begin_frame() can fail: the synchronous path always started the
// ImGui frame here, and the runtime's retry loop depends on that pairing.
if (imguiPolicy == ImGuiFramePolicy::Immediate) {
if (imgui::host_frames_active()) {
// The host starts its own ImGui frames (imgui::host_frame_begin).
} else if (imguiPolicy == ImGuiFramePolicy::Immediate) {
imgui::new_frame(window::get_window_size());
} else if (imguiNewFrameOwed != nullptr) {
*imguiNewFrameOwed = true;
@@ -1767,8 +1830,13 @@ struct SealedFrameContext {
std::optional<AuroraStereoFrame> stereoInput;
bool retainStereo = false;
imgui::StereoOverlay stereoOverlay;
imgui::HostFramePtr imguiFrame;
};
const ImDrawData* host_imgui_data(const SealedFrameContext& ctx) noexcept {
return ctx.imguiFrame ? imgui::host_frame_draw_data(*ctx.imguiFrame) : nullptr;
}
// Worker-owned scene state. A separate buffer generation check protects against
// synchronous EFB submissions overwriting the retained frame's GPU data.
struct RetainedStereoContext {
@@ -1827,8 +1895,11 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
// Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and
// a FIFO already drained into the recorded pass list.
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) {
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) {
ZoneScopedN("Seal frame");
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
// one frame; encode_sealed_frame resolves it on its last submission.
gfx::gpu_timing_begin_frame();
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
.label = "Redraw encoder",
};
@@ -1882,7 +1953,12 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u
ctx.presentSource = webgpu::current_present_source();
// ImGui draw lists are built once per frame and replayed by each slot's ImGui pass, which is why
// the next ImGui frame cannot start until the encode phase is done.
imgui::render_frame_data();
if (hostImGuiFrame) {
// The host closed its own ImGui frame and handed over a copy of the draw data.
ctx.imguiFrame = std::move(hostImGuiFrame);
} else {
imgui::render_frame_data();
}
// The headset panel's draw data follows the same rule on the host's side.
ctx.stereoOverlay = imgui::latch_stereo_overlay();
// Drop the sealed frame's lazy RAM-readback requests while the producer is still excluded; it
@@ -1974,7 +2050,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false);
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
presentationJobs.push_back({
.image = std::move(image),
.logicalFrame = ctx.logicalFrame,
@@ -1988,14 +2064,25 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo);
//
// On a headset an immersive frame's native render is never presented: the eyes replay the draws
// themselves and only sample the EFB copies it resolves. So it stops after the last pass that
// produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a
// 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame
// capture still gets the whole image.
int32_t nativeRenderLastPass = INT32_MAX;
if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() &&
g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) {
nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame);
}
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass);
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
// completion is harvested in gfx::after_submit, never waited on here.
gfx::efb_ram::encode_async_downloads(encoder);
if (!ctx.replayInterpolatedFrames) {
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
presentationJobs.push_back({
.image = std::move(image),
.logicalFrame = ctx.logicalFrame,
@@ -2035,7 +2122,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// showing. Black re-clears it below, once the eyes have taken their copy.
const bool virtualScreenNeedsMono = stereoOutput && !immersiveReplay;
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true,
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan);
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan, host_imgui_data(ctx));
if (stereoOutput) {
publish_stereo_screen_aspects(ctx.presentSource, finalImage->texture.size, immersiveReplay);
}
@@ -2057,7 +2144,8 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
stereo_overlay::composite_flat(encoder, output.view, output.size, eye);
}
if (mirrorPlan == MirrorPlan::Black && !headsetOnly) {
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black);
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black,
host_imgui_data(ctx));
}
}
if (stereoOutput) {
@@ -2070,7 +2158,9 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
.interpolated = false,
});
gfx::gpu_timing_end_frame(encoder);
submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
gfx::gpu_timing_after_submit();
// A group that finished encoding past its anchor slides forward by whole display periods, never
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
@@ -2180,11 +2270,22 @@ void record_frame_telemetry() {
{
// `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five
// seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this.
static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
static const bool fpsLog = [] {
const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
// The same switch turns on the per-pass GPU timestamps reported below the frame-rate line.
gfx::gpu_timing_set_enabled(on);
return on;
}();
if (fpsLog) {
static auto windowStart = std::chrono::steady_clock::now();
static uint32_t windowFrames = 0;
// Draw calls are what a recorded frame costs three times over (mono and both eyes), so they belong beside the
// GPU timings: an overlay that stops draws merging shows up here long before it shows up as a frame rate.
static uint64_t windowDraws = 0;
static uint64_t windowMerged = 0;
++windowFrames;
windowDraws += gfx::g_stats.drawCallCount;
windowMerged += gfx::g_stats.mergedDrawCallCount;
const auto now = std::chrono::steady_clock::now();
const std::chrono::duration<double> elapsed = now - windowStart;
if (elapsed.count() >= 5.0) {
@@ -2200,9 +2301,25 @@ void record_frame_telemetry() {
const double encode = msPerFrame(g_workerEncodeNs);
Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s); per frame the producer waited {:.2f} ms for "
"DONE and {:.2f} ms for SEALED; the worker spent {:.2f} ms sealing, {:.2f} ms waiting for the "
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding; {:.0f} draw calls a "
"frame ({:.0f} primitives merged away)",
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
permitWait, prepare, encode);
permitWait, prepare, encode,
static_cast<double>(windowDraws) / std::max(windowFrames, 1u),
static_cast<double>(windowMerged) / std::max(windowFrames, 1u));
windowDraws = 0;
windowMerged = 0;
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
Log.info("{}", gpuTiming);
}
if (const auto frameLog = g_frameLogCallback.load(std::memory_order_acquire)) {
char extra[512];
extra[0] = '\0';
frameLog(extra, sizeof(extra), elapsed.count(), windowFrames);
if (extra[0] != '\0') {
Log.info("{}", extra);
}
}
windowStart = now;
windowFrames = 0;
}
@@ -2214,7 +2331,7 @@ void record_frame_telemetry() {
// One complete frame-worker cycle. Desktop and headset interpolation both
// release the producer after sealing, before encoding their extra scene views.
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
const StereoSceneAnchor& sceneAnchor) noexcept {
ZoneScopedN("Frame worker cycle");
webgpu::fail_if_device_lost();
@@ -2226,7 +2343,7 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
auto stretchStarted = std::chrono::steady_clock::now();
{
std::lock_guard gpuLock(g_rendererGpuMutex);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
}
g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
stretchStarted = std::chrono::steady_clock::now();
@@ -2283,11 +2400,11 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
// Synchronous frame submission: seal, encode and present inline on the calling thread. Used when
// the frame worker is disabled (RenderDoc captures) and on the boot path.
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) noexcept {
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) noexcept {
ZoneScoped;
#ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost();
if (pumpEvents) {
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
window::pump_events();
}
gfx::SealedFrame sealedFrame;
@@ -2298,7 +2415,7 @@ void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
if (drainFifo) {
gx::fifo::drain();
}
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
}
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
@@ -2324,7 +2441,9 @@ bool begin_frame() noexcept {
ensure_frame_worker_started();
// SDL needs event pumping on the window-owning producer thread, and the worker passes
// pumpEvents=false, so keep it here even when the fast path returns early.
window::pump_events();
if (!g_hostEventPump.load(std::memory_order_acquire)) {
window::pump_events();
}
bool waitForSurfacePreparation = false;
#ifdef AURORA_ENABLE_GX
// A surface mutation can legitimately fail preparation, and optimistic success would let GX/ImGui
@@ -2374,7 +2493,7 @@ bool begin_frame() noexcept {
return prepared;
}
void end_frame(uint64_t contentTag) noexcept {
void end_frame(uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame) noexcept {
#ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost();
#endif
@@ -2385,7 +2504,7 @@ void end_frame(uint64_t contentTag) noexcept {
g_pendingStereoLocalPlayerCount = 1;
g_pendingSceneAnchor = {};
if (!frame_worker_requested()) {
end_frame_impl(true, true, contentTag, sceneAnchor);
end_frame_impl(true, true, contentTag, sceneAnchor, std::move(hostImGuiFrame));
return;
}
@@ -2407,6 +2526,7 @@ void end_frame(uint64_t contentTag) noexcept {
g_frameWorker.ready.store(false, std::memory_order_release);
g_frameWorker.contentTag = contentTag;
g_frameWorker.sceneAnchor = sceneAnchor;
g_frameWorker.hostImGuiFrame = std::move(hostImGuiFrame);
g_frameWorker.jobPending = true;
g_frameWorker.prepareAllowed = false;
}
@@ -2509,11 +2629,45 @@ AuroraInfo aurora_initialize(int argc, char* argv[], const AuroraConfig* config)
void aurora_shutdown() { aurora::shutdown(); }
const AuroraEvent* aurora_update() { return aurora::update(); }
bool aurora_begin_frame() { return aurora::begin_frame(); }
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN); }
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag); }
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN, {}); }
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag, {}); }
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame) {
aurora::imgui::HostFramePtr frame;
if (imguiFrame != nullptr) {
auto* holder = static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
frame = std::move(*holder);
delete holder;
}
aurora::end_frame(contentTag, std::move(frame));
}
void aurora_set_host_event_pump(bool hostPumps) {
aurora::g_hostEventPump.store(hostPumps, std::memory_order_release);
}
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback) {
aurora::g_frameLogCallback.store(callback, std::memory_order_release);
}
extern "C" void aurora_imgui_host_frame_begin(void) {
#ifdef AURORA_ENABLE_GX
// ImGui's WebGPU backend creates its device objects lazily from new_frame.
std::lock_guard gpuLock(aurora::g_rendererGpuMutex);
#endif
aurora::imgui::host_frame_begin(aurora::window::get_window_size());
}
extern "C" void* aurora_imgui_host_frame_end(void) { return new aurora::imgui::HostFramePtr(aurora::imgui::host_frame_end()); }
extern "C" void aurora_imgui_host_frame_release(void* imguiFrame) {
delete static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
}
void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]) {
aurora::set_stereo_scene_anchor(anchorFromScene);
}
void aurora_set_stereo_scene_anchor_scaled(const float anchorFromScene[12], float unitsPerMeter) {
aurora::set_stereo_scene_anchor(anchorFromScene);
uint32_t bits = 0;
std::memcpy(&bits, &unitsPerMeter, sizeof(bits));
if (aurora::g_pendingSceneAnchor.active && (bits & 0x7f800000u) != 0x7f800000u && unitsPerMeter > 0.f) {
aurora::g_pendingSceneAnchor.unitsPerMeter = unitsPerMeter;
}
}
void aurora_set_stereo_local_player_count(uint32_t count) {
aurora::g_pendingStereoLocalPlayerCount = count >= 1 && count <= 4 ? count : 1;
}
+58
View File
@@ -10,10 +10,68 @@
#include "../../gfx/common.hpp"
#include "../../gx/fifo.hpp"
#include "../../gx/native_wheel.hpp"
// GX-thread entry points for the VR native steering wheel (native_wheel.hpp).
// The runtime posts these in order with the frame's draws.
extern "C" void aurora_clear_native_wheel_vertices() {
// Clearing an empty set reports nothing, so a host that clears both before and after a frame's draws keeps
// that frame's count.
if (aurora::gx::nativeWheelArrays.empty()) return;
aurora::gx::fifo::drain();
aurora::gx::nativeWheelLastMatches.store(aurora::gx::nativeWheelMatches);
aurora::gx::nativeWheelMatches = 0;
aurora::gx::nativeWheelPreviousSources.clear();
for (const auto& array : aurora::gx::nativeWheelArrays)
aurora::gx::nativeWheelPreviousSources.push_back(array.source);
aurora::gx::native_wheel_report();
aurora::gx::nativeWheelArrays.clear();
aurora::gx::nativeWheelLastDecision = nullptr;
aurora::gx::nativeWheelLastDrawCommand = nullptr;
}
extern "C" uint32_t aurora_native_wheel_draw_count() { return aurora::gx::nativeWheelLastMatches.load(); }
extern "C" void aurora_set_native_wheel_vertices(const void* source, const void* replacement, uint32_t size,
const float* modelView) {
if (!source || !replacement || !modelView || !size || size > 65536) return;
aurora::gx::NativeWheelArray array;
array.source = source;
const auto* bytes = static_cast<const uint8_t*>(replacement);
array.bytes.assign(bytes, bytes + size);
std::memcpy(array.modelView.data(), modelView, sizeof(float) * 12);
aurora::gx::nativeWheelArrays.push_back(std::move(array));
// The vector may have moved its elements, and a set changes what a draw resolves to in any case, so the decision
// the merge test compares against is dropped rather than left pointing into the old storage.
aurora::gx::nativeWheelLastDecision = nullptr;
aurora::gx::nativeWheelLastDrawCommand = nullptr;
}
// Single definition for the `Log` that gx.hpp declares for this directory.
aurora::Module Log("aurora::gx");
namespace aurora::gx {
// Called as a set is cleared: a line at about half a second, five seconds and
// a minute of sets, with the matrices on the first report that saw a draw.
void native_wheel_report() {
auto& diagnostics=nativeWheelDiagnostics;
++diagnostics.sets;
++nativeWheelClears;
if(nativeWheelClears!=30 && nativeWheelClears!=300 && nativeWheelClears!=3600) return;
::Log.info("Native steering wheel: {} sets; draws binding a replaced array {} ({} with a larger range, {} outside a "
"set); matched {}; closest position matrix off by {} ({} matrices)",
diagnostics.sets,diagnostics.boundDraws,diagnostics.oversizeDraws,diagnostics.outsideDraws,
diagnostics.matchedDraws,diagnostics.bestError,diagnostics.bestIndexed?"indexed":"current");
if(diagnostics.boundDraws!=0 && nativeWheelReports++==0) {
const auto& m=diagnostics.bestMatrix;
const auto& e=diagnostics.expected;
::Log.info("Native steering wheel: closest [{} {} {} {} | {} {} {} {} | {} {} {} {}] expected [{} {} {} {} | {} {} "
"{} {} | {} {} {} {}]",
m[0],m[1],m[2],m[3],m[4],m[5],m[6],m[7],m[8],m[9],m[10],m[11],
e[0],e[1],e[2],e[3],e[4],e[5],e[6],e[7],e[8],e[9],e[10],e[11]);
}
diagnostics={};
}
} // namespace aurora::gx
static void GXWriteString(const char* label) {
auto length = strlen(label);
@@ -549,6 +549,13 @@ void GXCopyTex(void* dest, GXBool clear) {
.clearAlpha = true,
.clearDepth = false,
}),
.stereoPipeline = aurora::stereo_frame_provider_active() ? aurora::gfx::pipeline_ref(aurora::gfx::clear::PipelineConfig{
.msaaSamples = aurora::gfx::get_sample_count(),
.clearColor = false,
.clearAlpha = true,
.clearDepth = false,
.stereoStencil = true,
}) : 0,
.color = wgpu::Color{0.f, 0.f, 0.f, g_gxState.dstAlpha / 255.f},
});
}
+1 -1
View File
@@ -105,7 +105,7 @@ fn fs_main() -> FragmentOutput {
.targets = &colorTarget,
};
const wgpu::DepthStencilState depthStencil{
.format = g_graphicsConfig.depthFormat,
.format = config.stereoStencil ? wgpu::TextureFormat::Depth24PlusStencil8 : g_graphicsConfig.depthFormat,
.depthWriteEnabled = config.clearDepth,
.depthCompare = wgpu::CompareFunction::Always,
};
+3 -2
View File
@@ -7,6 +7,7 @@
namespace aurora::gfx::clear {
struct DrawData {
PipelineRef pipeline;
PipelineRef stereoPipeline = 0;
Range uniformRange;
wgpu::Color color;
float depth = 0.f;
@@ -18,14 +19,14 @@ struct DrawData {
ClipRect scissor{};
};
constexpr uint32_t ClearPipelineConfigVersion = 3;
constexpr uint32_t ClearPipelineConfigVersion = 4;
struct PipelineConfig {
uint32_t version = ClearPipelineConfigVersion;
uint32_t msaaSamples = 1;
bool clearColor = true;
bool clearAlpha = true;
bool clearDepth = true;
uint8_t _pad = 0;
bool stereoStencil = false;
};
static_assert(std::has_unique_object_representations_v<PipelineConfig>);
+299
View File
@@ -0,0 +1,299 @@
// SPDX-License-Identifier: GPL-3.0-or-later
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
//
// VR cockpit overlay: the synthetic steering wheel or handlebar (used when the
// vehicle's own wheel cannot be animated) and the tracked hands, drawn per eye
// in metres against the replayed scene's depth. See OPENXR.md, "Steering wheel
// and hand steering".
#pragma once
#include "common.hpp"
#include "../webgpu/gpu.hpp"
#include <array>
#include <atomic>
#include <cmath>
#include <cstring>
#include <memory>
#include <mutex>
#include <vector>
namespace aurora::gfx::cockpit {
using V = std::array<float, 3>;
using M = std::array<float, 12>;
inline V add(V a, V b) { return {a[0]+b[0], a[1]+b[1], a[2]+b[2]}; }
inline V sub(V a, V b) { return {a[0]-b[0], a[1]-b[1], a[2]-b[2]}; }
inline V mul(V a, float b) { return {a[0]*b, a[1]*b, a[2]*b}; }
inline float dot(V a, V b) { return a[0]*b[0]+a[1]*b[1]+a[2]*b[2]; }
inline V cross(V a, V b) { return {a[1]*b[2]-a[2]*b[1],a[2]*b[0]-a[0]*b[2],a[0]*b[1]-a[1]*b[0]}; }
inline V norm(V a) { return mul(a, 1/std::sqrt(std::max(dot(a,a), 1e-10f))); }
inline V point(const float* m, V p) {
return {m[0]*p[0]+m[1]*p[1]+m[2]*p[2]+m[3], m[4]*p[0]+m[5]*p[1]+m[6]*p[2]+m[7],
m[8]*p[0]+m[9]*p[1]+m[10]*p[2]+m[11]};
}
inline M identity() { return {1,0,0,0,0,1,0,0,0,0,1,0}; }
inline M compose(const M& a, const M& b) {
M result{};
for(int r=0;r<3;++r) {
for(int c=0;c<3;++c) for(int k=0;k<3;++k) result[r*4+c]+=a[r*4+k]*b[k*4+c];
result[r*4+3]=a[r*4+3];
for(int k=0;k<3;++k) result[r*4+3]+=a[r*4+k]*b[k*4+3];
}
return result;
}
inline M inverse(const M& m) {
M out=identity();
for(int r=0;r<3;++r) for(int c=0;c<3;++c) out[r*4+c]=m[c*4+r];
const auto p=point(out.data(), {-m[3],-m[7],-m[11]});
out[3]=p[0];out[7]=p[1];out[11]=p[2];return out;
}
inline M from_pose(const float* p) {
const float x=p[0],y=p[1],z=p[2],w=p[3];
return {1-2*(y*y+z*z),2*(x*y-z*w),2*(x*z+y*w),p[4],
2*(x*y+z*w),1-2*(x*x+z*z),2*(y*z-x*w),p[5],
2*(x*z-y*w),2*(y*z+x*w),1-2*(x*x+y*y),p[6]};
}
struct HandMesh {
std::vector<AuroraVRHandVertex> vertices;
std::vector<uint16_t> indices;
std::array<M,26> bind{}, inverseBind{};
std::array<int32_t,26> parents{};
};
inline std::mutex meshMutex;
inline std::array<std::shared_ptr<const HandMesh>,2> meshes;
struct Vertex { V position, color; };
inline void triangle(std::vector<Vertex>& vertices, V a, V b, V c, V color) {
const V normal=norm(cross(sub(b,a),sub(c,a)));
const float light=0.55f+0.45f*std::abs(dot(normal,norm({0.3f,0.8f,0.5f})));
color=mul(color,light);
vertices.insert(vertices.end(),{{a,color},{b,color},{c,color}});
}
inline void tube(std::vector<Vertex>& v, V a, V b, float radius, V color, int sides=8) {
const auto direction=norm(sub(b,a));
const auto u=norm(cross(direction,std::abs(direction[1])<0.9f?V{0,1,0}:V{1,0,0}));
const auto w=cross(direction,u);
for(int i=0;i<sides;++i) {
const float t=float(i)*6.2831853f/sides, t1=float(i+1)*6.2831853f/sides;
const V o=mul(add(mul(u,std::cos(t)),mul(w,std::sin(t))),radius);
const V p=mul(add(mul(u,std::cos(t1)),mul(w,std::sin(t1))),radius);
triangle(v,add(a,o),add(b,o),add(b,p),color);
triangle(v,add(a,o),add(b,p),add(a,p),color);
triangle(v,a,add(a,p),add(a,o),color);
triangle(v,b,add(b,o),add(b,p),color);
}
}
inline void ellipsoid(std::vector<Vertex>& vertices,V center,V radii,V color) {
const auto surface=[&](int ring,int segment) {
const float latitude=float(ring)*3.14159265f/6,longitude=float(segment)*6.2831853f/12;
return add(center,{radii[0]*std::sin(latitude)*std::cos(longitude),radii[1]*std::cos(latitude),
radii[2]*std::sin(latitude)*std::sin(longitude)});
};
for(int ring=0;ring<6;++ring) for(int segment=0;segment<12;++segment) {
const auto a=surface(ring,segment),b=surface(ring+1,segment),c=surface(ring+1,segment+1),d=surface(ring,segment+1);
if(ring>0) triangle(vertices,a,b,d,color);
if(ring<5) triangle(vertices,b,c,d,color);
}
}
// Rounded palm and individually articulated fingers, in the controller's grip
// space as OpenXR defines it: the origin is the palm centroid, -Z runs up the
// tube the curled fingers form (little finger towards thumb), and +X is normal
// to the palm - *away* from it on the left hand, *into* it on the right. That
// asymmetry is what makes both grips carry the same orientation when the hands
// hold a wheel symmetrically, so the fingers run along -Y on both, and it is
// the geometry across the palm that mirrors: fingers close towards +X on the
// left hand and -X on the right, with the thumb on the same side. Building the
// fingers on any other axis bends them out of the back of the hand (seen on a
// Quest 3 on 2026-09-22) or, for the right hand alone, points them at the
// player (seen on the PC on 2026-09-23).
inline void glove(std::vector<Vertex>& v, const AuroraCockpitHand& hand, int side) {
const size_t start=v.size();
const V white{0.91f,0.95f,1.0f};
const float palm=side==0?1.0f:-1.0f; // hand 0 is the left one
const float curl=std::clamp(hand.held?0.85f:hand.squeeze,0.0f,1.0f);
// Thin through the palm's normal, a little wider across the knuckles than
// the palm is long.
ellipsoid(v,{0,0,0},{0.018f,0.043f,0.041f},white);
for(int finger=0;finger<4;++finger) {
// Index finger nearest the thumb (-Z), little finger last.
V a{0.0f,-0.030f,-0.025f+finger*0.017f};
const float length=finger==0||finger==3?0.021f:0.026f;
for(int joint=0;joint<3;++joint) {
const float angle=curl*(0.55f+joint*0.8f);
V b=add(a,{palm*std::sin(angle)*length,-std::cos(angle)*length,0.0f});
tube(v,a,b,0.008f,white);
ellipsoid(v,b,{0.008f,0.008f,0.008f},white);a=b;
}
}
// Thumb: out of the palm's thumb side, closing across the fingers.
const V thumbKnuckle{palm*0.026f,-0.034f,-0.030f};
tube(v,{palm*0.010f,-0.012f,-0.034f},thumbKnuckle,0.010f,white);
tube(v,thumbKnuckle,{palm*(0.030f+0.014f*curl),-(0.052f-0.016f*curl),-0.020f},0.009f,white);
for(size_t i=start;i<v.size();++i) v[i].position=point(hand.seatFromGrip,v[i].position);
}
inline void runtime_hand(std::vector<Vertex>& out, const AuroraCockpitHand& hand, const HandMesh& mesh) {
std::array<M,26> posed{}, skin{};
std::array<bool,26> done{};
const float curl=std::clamp(hand.held?0.85f:hand.squeeze,0.0f,1.0f);
// Bind hierarchy is supplied by the runtime. Root and wrist stay rigid;
// finger joints curl locally when controllers provide squeeze input.
for(int pass=0;pass<26;++pass) for(int j=0;j<26;++j) {
if(done[j]) continue;
const int parent=mesh.parents[j];
if(parent>=0&&parent<26&&!done[parent]) continue;
M local=parent>=0&&parent<26?compose(mesh.inverseBind[parent],mesh.bind[j]):mesh.bind[j];
const bool fingerJoint=j>=2 && j!=6 && j!=11 && j!=16 && j!=21;
if(fingerJoint) {
// OpenXR joints point -Z toward the fingertip and +Y out of the back
// of the hand. Flexion is therefore negative about local X, for both
// hands; positive angles bend the fingers backward on runtime meshes.
const float a=-curl*(j<6?0.3f:0.75f),c=std::cos(a),s=std::sin(a);
local=compose(local,M{1,0,0,0,0,c,-s,0,0,s,c,0});
}
posed[j]=parent>=0&&parent<26?compose(posed[parent],local):local;
skin[j]=compose(mesh.inverseBind[1],compose(posed[j],mesh.inverseBind[j]));
done[j]=true;
}
std::vector<V> points(mesh.vertices.size());
for(size_t i=0;i<points.size();++i) {
const auto& v=mesh.vertices[i]; V p{}; float total=0;
for(int w=0;w<4;++w) if(v.joints[w]>=0&&v.joints[w]<26&&done[v.joints[w]]&&v.weights[w]>0) {
p=add(p,mul(point(skin[v.joints[w]].data(),{v.position[0],v.position[1],v.position[2]}),v.weights[w]));
total+=v.weights[w];
}
if(total>0) p=mul(p,1/total);
p=add(p,{0,0,0.04f}); // wrist behind the controller grip/palm origin.
points[i]=point(hand.seatFromGrip,p);
}
for(size_t i=0;i+2<mesh.indices.size();i+=3)
triangle(out,points[mesh.indices[i]],points[mesh.indices[i+1]],points[mesh.indices[i+2]],{0.91f,0.95f,1.0f});
}
inline void build_geometry(const AuroraCockpit& cockpit, std::vector<Vertex>& vertices) {
vertices.clear();vertices.reserve(12000);
// The visible radius and position must match runtime/vr/steering_wheel.h.
if (!cockpit.nativeWheel && cockpit.bike) {
const float c=std::cos(cockpit.wheelAngle),s=std::sin(cockpit.wheelAngle);
const auto barPoint=[&](float x,float y,float z) {
return point(cockpit.seatFromHandlebar,{c*x+s*y,-s*x+c*y,z});
};
const float radius=cockpit.handlebarRadius;
tube(vertices,barPoint(-radius,0,0),barPoint(radius,0,0),0.013f,{0.45f,0.48f,0.52f});
for(float side:{-1.0f,1.0f})
tube(vertices,barPoint(side*std::max(radius-0.10f,0.0f),0,0),barPoint(side*radius,0,0),0.024f,{0.12f,0.18f,0.19f});
tube(vertices,barPoint(0,0,-0.13f),barPoint(0,0,0),0.023f,{0.12f,0.65f,0.61f});
} else if (!cockpit.nativeWheel) {
const auto rim=[&](float angle) -> V { return {0.18f*std::cos(angle),-0.30f+0.18f*std::sin(angle),-0.42f}; };
for(int i=0;i<64;++i) {
const float angle=float(i)*6.2831853f/64-cockpit.wheelAngle;
const V color=i>=15&&i<=17?V{0.2f,0.9f,0.8f}:V{0.14f,0.17f,0.20f};
tube(vertices,rim(angle),rim(angle+6.2831853f/64),0.016f,color,6);
}
for(float a : {0.0f,3.14159265f,4.71238898f})
tube(vertices,{0,-0.30f,-0.42f},rim(a-cockpit.wheelAngle),0.011f,{0.45f,0.48f,0.52f});
tube(vertices,{0,-0.30f,-0.445f},{0,-0.30f,-0.395f},0.035f,{0.12f,0.65f,0.61f},16);
}
std::array<std::shared_ptr<const HandMesh>,2> current;
{ std::lock_guard lock(meshMutex);current=meshes; }
for(int side=0;side<2;++side) if(cockpit.hands[side].tracked) {
if(current[side]) runtime_hand(vertices,cockpit.hands[side],*current[side]);
else glove(vertices,cockpit.hands[side],side);
}
}
inline std::vector<Vertex> geometry(const AuroraCockpit& cockpit) {
std::vector<Vertex> result;build_geometry(cockpit,result);return result;
}
inline std::atomic<uint64_t> meshRevision{1};
inline std::vector<Vertex> frameVertices;
inline AuroraCockpit cachedCockpit{};
inline uint64_t cachedMeshRevision=0;
inline wgpu::RenderPipeline pipeline;
struct SceneDepth {
float z=0, constant=0;
bool valid=false;
};
inline uint32_t pipelineSamples=0;
inline bool pipelineReversedDepth=false;
inline wgpu::TextureFormat pipelineFormat{}, pipelineDepthFormat{};
inline std::array<wgpu::Buffer,2> vertexBuffers;
inline std::array<uint64_t,2> vertexCapacity{};
inline void shutdown() { pipeline=nullptr;pipelineSamples=0;vertexBuffers={};vertexCapacity={};cachedMeshRevision=0;frameVertices.clear(); }
inline void render(wgpu::CommandEncoder& cmd,const StereoReplayFrame& frame,uint32_t eye,SceneDepth sceneDepth={},
const wgpu::RenderPassEncoder* existingPass=nullptr) {
if(!frame.cockpit.active || !sceneDepth.valid) return;
using namespace webgpu;
const auto& target=frame.eyes[eye].target;
const auto format=g_graphicsConfig.surfaceConfiguration.format;
// The guest can reverse its viewport depth independently of Aurora's
// global reversed-Z convention. The final 1/d coefficient is authoritative.
const bool reversedDepth=sceneDepth.constant>0;
if(!pipeline||pipelineSamples!=target.msaaSamples||pipelineFormat!=format||pipelineReversedDepth!=reversedDepth||pipelineDepthFormat!=target.depthFormat) {
wgpu::ShaderSourceWGSL source{};
source.code=R"(
struct Out { @builtin(position) position: vec4f, @location(0) color: vec3f };
@vertex fn vs(@location(0) position: vec4f, @location(1) color: vec3f) -> Out {
var o: Out; o.position=position; o.color=color; return o;
}
@fragment fn fs(i: Out) -> @location(0) vec4f { return vec4f(i.color,1); }
)";
wgpu::ShaderModuleDescriptor md{};md.nextInChain=&source;md.label="VR cockpit hands and wheel";
auto shader=g_device.CreateShaderModule(&md);
const wgpu::VertexAttribute attrs[]={{.format=wgpu::VertexFormat::Float32x4,.offset=0,.shaderLocation=0},
{.format=wgpu::VertexFormat::Float32x3,.offset=16,.shaderLocation=1}};
const wgpu::VertexBufferLayout layout{.arrayStride=28,.attributeCount=2,.attributes=attrs};
const wgpu::ColorTargetState color{.format=format};
const wgpu::FragmentState fragment{.module=shader,.entryPoint="fs",.targetCount=1,.targets=&color};
const bool stencil=target.depthFormat==wgpu::TextureFormat::Depth24PlusStencil8;
const wgpu::StencilFaceState mark{.compare=wgpu::CompareFunction::Always,
.passOp=stencil?wgpu::StencilOperation::Replace:wgpu::StencilOperation::Keep};
const wgpu::DepthStencilState depth{.format=target.depthFormat,.depthWriteEnabled=true,
.depthCompare=reversedDepth?wgpu::CompareFunction::GreaterEqual:wgpu::CompareFunction::LessEqual,
.stencilFront=mark,.stencilBack=mark,.stencilReadMask=1,.stencilWriteMask=stencil?1u:0u};
wgpu::RenderPipelineDescriptor desc{};desc.label="VR cockpit";
desc.vertex={.module=shader,.entryPoint="vs",.bufferCount=1,.buffers=&layout};
desc.fragment=&fragment;desc.depthStencil=&depth;desc.multisample.count=target.msaaSamples;
desc.primitive.topology=wgpu::PrimitiveTopology::TriangleList;
pipeline=g_device.CreateRenderPipeline(&desc);pipelineSamples=target.msaaSamples;pipelineFormat=format;
pipelineReversedDepth=reversedDepth;pipelineDepthFormat=target.depthFormat;
}
const auto revision=meshRevision.load();
if(cachedMeshRevision!=revision || std::memcmp(&cachedCockpit,&frame.cockpit,sizeof(AuroraCockpit))!=0) {
build_geometry(frame.cockpit,frameVertices);
cachedCockpit=frame.cockpit;cachedMeshRevision=revision;
}
const auto& vertices=frameVertices;
if(vertices.empty()) return;
struct ClipVertex { float p[4]; V color; };
static std::vector<ClipVertex> clip;
clip.resize(vertices.size());
const auto& projection=frame.eyes[eye].projection;
for(size_t i=0;i<clip.size();++i) {
const auto p=point(frame.cockpit.eyeFromSeat[eye],vertices[i].position);
// The original race near plane can sit beyond a close hand. Keep that
// hand at the nearest representable depth instead of clipping it away.
const float z=sceneDepth.z*p[2]+sceneDepth.constant/std::max(frame.cockpit.unitsPerMeter,0.001f);
clip[i]={{projection.m0[0]*p[0]+projection.m0[2]*p[2],projection.m1[1]*p[1]+projection.m1[2]*p[2],
std::clamp(z,0.0f,std::max(-p[2],0.0f)),-p[2]},vertices[i].color};
}
const uint64_t bytes=clip.size()*sizeof(ClipVertex);
if (!vertexBuffers[eye] || vertexCapacity[eye]<bytes) {
vertexCapacity[eye]=(bytes+65535)&~uint64_t(65535);
const wgpu::BufferDescriptor bd{.label="VR cockpit vertices",.usage=wgpu::BufferUsage::Vertex|wgpu::BufferUsage::CopyDst,
.size=vertexCapacity[eye]};
vertexBuffers[eye]=g_device.CreateBuffer(&bd);
}
auto& buffer=vertexBuffers[eye];
g_queue.WriteBuffer(buffer,0,clip.data(),bytes);
const wgpu::RenderPassColorAttachment attachment{.view=target.colorView,.resolveTarget=target.resolveView,
.loadOp=wgpu::LoadOp::Load,.storeOp=wgpu::StoreOp::Store};
const wgpu::RenderPassDepthStencilAttachment depth{.view=target.depthView,.depthLoadOp=wgpu::LoadOp::Load,
.depthStoreOp=wgpu::StoreOp::Store,.depthClearValue=1.0f,
.stencilLoadOp=target.depthFormat==wgpu::TextureFormat::Depth24PlusStencil8?wgpu::LoadOp::Load:wgpu::LoadOp::Undefined,
.stencilStoreOp=target.depthFormat==wgpu::TextureFormat::Depth24PlusStencil8?wgpu::StoreOp::Store:wgpu::StoreOp::Undefined};
const wgpu::RenderPassDescriptor pd{.label="VR cockpit overlay",.colorAttachmentCount=1,.colorAttachments=&attachment,.depthStencilAttachment=&depth};
auto pass=existingPass?*existingPass:cmd.BeginRenderPass(&pd);
pass.SetViewport(0,0,float(target.size.width),float(target.size.height),0,1);
pass.SetScissorRect(0,0,target.size.width,target.size.height);
// Mark only depth-visible samples; later virtual-screen draws test for zero.
pass.SetStencilReference(1);
pass.SetPipeline(pipeline);pass.SetVertexBuffer(0,buffer);pass.Draw(clip.size());
pass.SetStencilReference(0);
if(!existingPass) pass.End();
}
} // namespace aurora::gfx::cockpit
+347 -2
View File
@@ -9,6 +9,7 @@
#include "../gx/pipeline.hpp"
#include "pipeline_cache.hpp"
#include "stereo_replay.hpp"
#include "cockpit.hpp"
#include "tex_copy_conv.hpp"
#include "tex_palette_conv.hpp"
#include "texture_replacement.hpp"
@@ -158,6 +159,9 @@ uint32_t g_mergedDrawCallCount = 0;
using CommandList = std::vector<Command>;
struct RenderPass {
// The world depth mapping of this pass's last full-view perspective draw, for
// the VR cockpit overlay (set by prepare_stereo_replay_uniforms).
cockpit::SceneDepth cockpitDepth{};
wgpu::TextureView colorView;
wgpu::TextureView resolveView; // MSAA resolve target; null if msaaSamples == 1
wgpu::TextureView depthView;
@@ -733,6 +737,13 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
.clearAlpha = clearAlpha,
.clearDepth = clearDepth,
}),
.stereoPipeline = aurora::stereo_frame_provider_active() ? pipeline_ref(clear::PipelineConfig{
.msaaSamples = msaaSamples,
.clearColor = clearColor,
.clearAlpha = clearAlpha,
.clearDepth = clearDepth,
.stereoStencil = true,
}) : 0,
.color =
wgpu::Color{
.r = clearColorValue.x(),
@@ -1057,6 +1068,7 @@ void initialize() {
}
void shutdown() {
cockpit::shutdown();
shutdown_pipeline_cache();
gx::clear_shader_module_cache();
efb_ram::shutdown();
@@ -1512,6 +1524,7 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame,
std::array<uint8_t, gx::MaxUniformSize> sourceUniform;
std::array<uint8_t, gx::MaxUniformSize> eyeUniform;
for (auto& pass : g_renderPasses) {
pass.cockpitDepth = {};
if (!pass.efbTarget) {
continue;
}
@@ -1541,6 +1554,19 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame,
std::memcpy(sourceUniform.data(), g_uniforms.data() + draw.uniformRange.offset, draw.uniformRange.size);
Mat4x4<float> gameProjection;
std::memcpy(&gameProjection, sourceUniform.data() + layout.projectionOffset, sizeof(gameProjection));
// The VR cockpit overlay (hands, synthetic wheel) is drawn in metres and
// depth-tested against the world, so it needs the world's own depth
// mapping: the backend depth row of a full-view world draw, with this
// viewport's depth range folded in because the overlay draws with 0..1.
// Camera-attached effects share the camera's projection, so any full-view
// perspective draw describes the same mapping.
if (layout.perspective && !layout.nativeEfbEffect && gameProjection.m2[3] != 0.0f &&
drawViewport.width >= displayRegion.width * 0.9f && drawViewport.height >= displayRegion.height * 0.9f) {
const auto row = stereo_replay::backend_ndc_depth_row(gameProjection);
const float low = std::clamp(std::min(drawViewport.znear, drawViewport.zfar), 0.f, 1.f);
const float high = std::clamp(std::max(drawViewport.znear, drawViewport.zfar), 0.f, 1.f);
pass.cockpitDepth = {row[2] * (high - low) - low, row[3] * (high - low), true};
}
// Only a genuinely affine projection carries its NDC position in its clip
// position, which is what the virtual screen reprojection consumes. GX
// tracks the projection type separately from the matrix, so a 2D draw
@@ -1724,12 +1750,22 @@ struct RenderInvocation {
uint32_t localPlayerCount = 1;
// Inclusive index of the last pass to replay; -1 replays every pass.
int32_t replayLastPass = -1;
// Inclusive index of the last pass that does render work; texture bakes still run for the
// passes after it. See last_pass_feeding_replay.
int32_t renderLastPass = INT32_MAX;
bool finalize = true;
bool replayOnlyEfb = false;
bool skipCopyClears = false;
bool encodeTextureBakes = true;
bool encodeResolves = true;
bool captureDepth = true;
// VR cockpit overlay, drawn inside the scene's pass just before the first
// virtual-screen draw so the 2D layer's depth cannot hide it (see render_stereo_eye).
const StereoReplayFrame* cockpitFrame = nullptr;
wgpu::CommandEncoder* cockpitEncoder = nullptr;
cockpit::SceneDepth cockpitDepth{};
bool* cockpitDrawn = nullptr;
bool* sceneDrawn = nullptr;
};
static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vector<RenderPass>& passes, u32 idx,
@@ -1740,6 +1776,9 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
ZoneScoped;
// Palette conversions, MSAA resolves and EFB copies depend on sealed frame state, not on the
// interpolation weight, so encode them on the native render and let replay slots sample them.
// Eye textures are reused; discard the previous frame's mask, then retain it
// across guest passes even if the HUD clears or replaces guest depth.
bool stencilInitialized = false;
for (u32 i = 0; i < renderPasses.size(); ++i) {
const auto& passInfo = renderPasses[i];
if (invocation.replayLastPass >= 0 && i > static_cast<u32>(invocation.replayLastPass)) {
@@ -1755,6 +1794,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
tex_palette_conv::run(cmd, conv);
}
}
if (static_cast<int32_t>(i) > invocation.renderLastPass) {
// Nothing after the last replay-feeding resolve is shown or sampled on a headset; the
// bakes above are all these passes owe the eye replays.
continue;
}
const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty();
if (i == renderPasses.size() - 1) {
ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target");
@@ -1787,19 +1831,30 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
},
},
};
const bool stereoStencil = overrideTarget &&
invocation.target->depthFormat == wgpu::TextureFormat::Depth24PlusStencil8;
const wgpu::RenderPassDepthStencilAttachment depthStencilAttachment{
.view = depthView,
.depthLoadOp = passInfo.clearDepth && !dropCopyClear ? wgpu::LoadOp::Clear : wgpu::LoadOp::Load,
.depthStoreOp = wgpu::StoreOp::Store,
.depthClearValue = passInfo.clearDepthValue,
.stencilLoadOp = stereoStencil ? (stencilInitialized ? wgpu::LoadOp::Load : wgpu::LoadOp::Clear) : wgpu::LoadOp::Undefined,
.stencilStoreOp = stereoStencil ? wgpu::StoreOp::Store : wgpu::StoreOp::Undefined,
.stencilClearValue = 0,
};
const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft
: invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight
: invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated
: GpuTimingCategory::Mono;
const wgpu::RenderPassDescriptor renderPassDescriptor{
.label = render_pass_label(i),
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.depthStencilAttachment = &depthStencilAttachment,
.timestampWrites = gpu_timing_pass(timingCategory),
};
if (stereoStencil) stencilInitialized = true;
auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
render_pass_impl(pass, renderPasses, i, invocation);
pass.End();
@@ -1908,15 +1963,28 @@ void seal_frame(SealedFrame& out) noexcept {
g_currentRenderPass = UINT32_MAX;
}
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize,
int32_t nativeRenderLastPass) {
render_impl(frame.data().passes, cmd,
RenderInvocation{
.interpolatedFrame = interpolatedFrame,
.renderLastPass = nativeRenderLastPass,
.finalize = finalize,
.encodeTextureBakes = interpolatedFrame < 0,
});
}
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept {
const auto& passes = frame.data().passes;
int32_t last = -1;
for (size_t i = 0; i < passes.size(); ++i) {
if (passes[i].resolveTarget && !passes[i].displayCopyResolve) {
last = static_cast<int32_t>(i);
}
}
return last;
}
bool has_late_stereo_replay(const SealedFrame& frame) noexcept {
const auto& data = frame.data().stereo;
return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) &&
@@ -1999,6 +2067,18 @@ void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, const Ster
// The eye is a fresh per-frame attachment, not the reused EFB, so replaying
// past that copy blanks the very image the game presented.
const int32_t lastPass = get_stereo_stop_at_display_copy() ? displaySource.lastDisplayCopyPass : -1;
cockpit::SceneDepth cockpitDepth{};
for (size_t i = 0; i < frame.data().passes.size(); ++i) {
if (lastPass >= 0 && i > static_cast<size_t>(lastPass)) {
break;
}
if (frame.data().passes[i].cockpitDepth.valid) {
cockpitDepth = frame.data().passes[i].cockpitDepth;
}
}
bool cockpitDrawn = false;
bool sceneDrawn = false;
const bool cockpitActive = stereoFrame.cockpit.active && cockpitDepth.valid;
render_impl(frame.data().passes, cmd,
RenderInvocation{
.stereoEye = eye,
@@ -2012,7 +2092,17 @@ void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, const Ster
.encodeTextureBakes = false,
.encodeResolves = false,
.captureDepth = false,
.cockpitFrame = cockpitActive ? &stereoFrame : nullptr,
.cockpitEncoder = &cmd,
.cockpitDepth = cockpitDepth,
.cockpitDrawn = &cockpitDrawn,
.sceneDrawn = &sceneDrawn,
});
// A frame without a virtual-screen draw after its world still gets the
// overlay, in a pass of its own over the finished eye.
if (cockpitActive && !cockpitDrawn) {
cockpit::render(cmd, stereoFrame, eye, cockpitDepth);
}
}
void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
@@ -2028,6 +2118,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize)
}
}
// --- Per-pass GPU timing (see common.hpp) -------------------------------------------------------
namespace {
constexpr uint32_t kGpuTimingSlots = 4;
constexpr uint32_t kGpuTimingPairs = 62;
constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs;
struct GpuTimingSlot {
wgpu::QuerySet querySet;
wgpu::Buffer resolve;
wgpu::Buffer readback;
std::array<wgpu::PassTimestampWrites, kGpuTimingPairs> writes{};
std::array<GpuTimingCategory, kGpuTimingPairs> categories{};
uint32_t pairs = 0;
bool open = false; // between the frame's begin and end
bool reading = false; // readback in flight or mapped
bool mapped = false; // the callback ran; the encoding thread unmaps on reuse
};
std::atomic<bool> g_gpuTimingEnabled{false};
std::array<GpuTimingSlot, kGpuTimingSlots> g_gpuTimingSlots;
uint32_t g_gpuTimingNextSlot = 0;
int32_t g_gpuTimingCurrent = -1;
bool g_gpuTimingReady = false;
// Guards the totals below and every slot's reading/mapped flags: the map callback may run on
// whichever thread processes Dawn's events.
std::mutex g_gpuTimingMutex;
std::array<uint64_t, static_cast<size_t>(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{};
uint64_t g_gpuTimingSpanNs = 0;
uint32_t g_gpuTimingFrames = 0;
uint32_t g_gpuTimingSkipped = 0;
bool gpu_timing_create_slots() {
if (g_gpuTimingReady) {
return true;
}
if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) {
return false;
}
for (auto& slot : g_gpuTimingSlots) {
const wgpu::QuerySetDescriptor querySetDescriptor{
.label = "GPU timing queries",
.type = wgpu::QueryType::Timestamp,
.count = kGpuTimingQueries,
};
slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor);
const wgpu::BufferDescriptor resolveDescriptor{
.label = "GPU timing resolve",
.usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor);
const wgpu::BufferDescriptor readbackDescriptor{
.label = "GPU timing readback",
.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst,
.size = kGpuTimingQueries * sizeof(uint64_t),
};
slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor);
}
g_gpuTimingReady = true;
return true;
}
} // namespace
void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); }
bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); }
void gpu_timing_begin_frame() noexcept {
g_gpuTimingCurrent = -1;
if (!gpu_timing_enabled() || !gpu_timing_create_slots()) {
return;
}
const uint32_t index = g_gpuTimingNextSlot;
g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots;
auto& slot = g_gpuTimingSlots[index];
{
std::lock_guard lock(g_gpuTimingMutex);
if (slot.reading && !slot.mapped) {
++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed
return;
}
if (slot.mapped) {
slot.readback.Unmap();
slot.mapped = false;
}
slot.reading = false;
}
slot.pairs = 0;
slot.open = true;
g_gpuTimingCurrent = static_cast<int32_t>(index);
}
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept {
if (g_gpuTimingCurrent < 0) {
return nullptr;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
if (!slot.open || slot.pairs >= kGpuTimingPairs) {
return nullptr;
}
const uint32_t i = slot.pairs++;
slot.writes[i] = wgpu::PassTimestampWrites{
.querySet = slot.querySet,
.beginningOfPassWriteIndex = 2 * i,
.endOfPassWriteIndex = 2 * i + 1,
};
slot.categories[i] = category;
return &slot.writes[i];
}
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
slot.open = false;
if (slot.pairs == 0) {
g_gpuTimingCurrent = -1;
return;
}
const uint32_t queries = 2 * slot.pairs;
encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0);
encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t));
}
void gpu_timing_after_submit() noexcept {
if (g_gpuTimingCurrent < 0) {
return;
}
const uint32_t index = static_cast<uint32_t>(g_gpuTimingCurrent);
g_gpuTimingCurrent = -1;
auto& slot = g_gpuTimingSlots[index];
const uint32_t pairs = slot.pairs;
{
std::lock_guard lock(g_gpuTimingMutex);
slot.reading = true;
slot.mapped = false;
}
slot.readback.MapAsync(
wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous,
[index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) {
auto& slot = g_gpuTimingSlots[index];
std::lock_guard lock(g_gpuTimingMutex);
if (status != wgpu::MapAsyncStatus::Success) {
slot.reading = false;
return;
}
const auto* stamps =
static_cast<const uint64_t*>(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t)));
if (stamps != nullptr) {
uint64_t first = UINT64_MAX;
uint64_t last = 0;
for (uint32_t i = 0; i < pairs; ++i) {
const uint64_t begin = stamps[2 * i];
const uint64_t end = stamps[2 * i + 1];
if (end < begin) {
continue;
}
g_gpuTimingTotalsNs[static_cast<size_t>(slot.categories[i])] += end - begin;
first = std::min(first, begin);
last = std::max(last, end);
}
if (last > first) {
g_gpuTimingSpanNs += last - first;
}
++g_gpuTimingFrames;
}
slot.mapped = true;
});
}
std::string gpu_timing_report() {
std::lock_guard lock(g_gpuTimingMutex);
if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) {
return {};
}
static constexpr std::array<const char*, static_cast<size_t>(GpuTimingCategory::Count)> kNames{
"mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"};
std::string text;
if (g_gpuTimingFrames != 0) {
const double frames = g_gpuTimingFrames;
uint64_t sum = 0;
text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames,
static_cast<double>(g_gpuTimingSpanNs) / 1e6 / frames);
for (size_t i = 0; i < kNames.size(); ++i) {
if (g_gpuTimingTotalsNs[i] == 0) {
continue;
}
sum += g_gpuTimingTotalsNs[i];
text += fmt::format(" {}={:.2f}", kNames[i], static_cast<double>(g_gpuTimingTotalsNs[i]) / 1e6 / frames);
}
const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0;
text += fmt::format(" between-passes={:.2f}", static_cast<double>(between) / 1e6 / frames);
}
if (g_gpuTimingSkipped != 0) {
text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped);
}
g_gpuTimingTotalsNs.fill(0);
g_gpuTimingSpanNs = 0;
g_gpuTimingFrames = 0;
g_gpuTimingSkipped = 0;
return text;
}
void after_submit() noexcept {
depth_peek::after_submit();
efb_ram::after_submit();
@@ -2223,6 +2516,24 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
draw.gx.interpolatedUniformRanges[invocation.interpolatedFrame].size != 0) {
uniformOverride = &draw.gx.interpolatedUniformRanges[invocation.interpolatedFrame];
}
// Draw against world depth and mark visible cockpit samples before HUD
// depth replaces it. The screen pipelines reject those stencil samples.
if (invocation.cockpitFrame != nullptr && overrideTarget) {
if (draw.gx.uniformReplayLayout.perspective) {
*invocation.sceneDrawn = true;
}
if (virtualScreenDraw && *invocation.sceneDrawn && !*invocation.cockpitDrawn) {
cockpit::render(*invocation.cockpitEncoder, *invocation.cockpitFrame, invocation.stereoEye,
invocation.cockpitDepth, &pass);
*invocation.cockpitDrawn = true;
encodeState = {};
encodeState.boundTextureBindGroup = gx::g_emptyTextureBindGroup.Get();
pass.SetBindGroup(0, g_staticBindGroup);
pass.SetBindGroup(2, gx::g_emptyTextureBindGroup);
scissorStateKnown = false;
viewportStateKnown = false;
}
}
// Such a draw no longer lands where the game aimed it, while the
// recorded scissor still describes the rectangle it occupied on the flat
// frame (Mario Kart clips the item roulette that way). Honouring that
@@ -2240,10 +2551,15 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
apply_viewport(fullEyeDraw);
}
gx::render(draw.gx, pass, encodeState, renderPasses[idx].requireReadyPipelines, uniformOverride,
virtualScreenDraw ? draw.gx.exactScreenDepthPipeline : 0);
overrideTarget && invocation.target->depthFormat == wgpu::TextureFormat::Depth24PlusStencil8
? (virtualScreenDraw ? draw.gx.stereoScreenPipeline : draw.gx.stereoPipeline)
: (virtualScreenDraw ? draw.gx.exactScreenDepthPipeline : 0));
} break;
case ShaderType::Clear: {
auto clearDraw = draw.clear;
if (overrideTarget && invocation.target->depthFormat == wgpu::TextureFormat::Depth24PlusStencil8) {
clearDraw.pipeline = clearDraw.stereoPipeline;
}
if (multiplayer) {
const auto& sc = clearDraw.scissor;
if (clearDraw.copyClear ||
@@ -2478,3 +2794,32 @@ void aurora_pop_debug_group() {
}
const AuroraStats* aurora_get_stats() { return &aurora::gfx::g_stats; }
void aurora_set_vr_hand_mesh(uint32_t hand, const AuroraVRHandVertex* vertices, uint32_t vertexCount,
const uint16_t* indices, uint32_t indexCount, const float* bindPoses,
const int32_t* parents, uint32_t jointCount) {
using namespace aurora::gfx::cockpit;
if (hand >= 2) {
return;
}
std::shared_ptr<HandMesh> mesh;
if (vertices && indices && bindPoses && parents && jointCount == 26 && vertexCount > 0 && vertexCount <= 65535 &&
indexCount <= 100000 && indexCount % 3 == 0) {
for (uint32_t i = 0; i < indexCount; ++i) {
if (indices[i] >= vertexCount) {
return;
}
}
mesh = std::make_shared<HandMesh>();
mesh->vertices.assign(vertices, vertices + vertexCount);
mesh->indices.assign(indices, indices + indexCount);
for (int j = 0; j < 26; ++j) {
mesh->bind[j] = from_pose(bindPoses + j * 7);
mesh->inverseBind[j] = inverse(mesh->bind[j]);
mesh->parents[j] = parents[j];
}
}
std::lock_guard lock(meshMutex);
meshes[hand] = std::move(mesh);
++meshRevision;
}
+45 -1
View File
@@ -8,6 +8,7 @@
#include <cstring>
#include <array>
#include <memory>
#include <string>
#include <type_traits>
#include <utility>
@@ -290,6 +291,7 @@ struct ReplayTarget {
wgpu::TextureView copySourceDepthView;
wgpu::Extent3D size{};
uint32_t msaaSamples = 1;
wgpu::TextureFormat depthFormat = wgpu::TextureFormat::Depth32Float;
};
struct StereoReplayEye {
@@ -313,6 +315,8 @@ struct StereoReplayEye {
struct StereoReplayFrame {
std::array<StereoReplayEye, AURORA_STEREO_EYE_COUNT> eyes;
// VR hands and synthetic wheel, drawn per eye after the world (gfx/cockpit.hpp).
AuroraCockpit cockpit{};
};
void end_frame(const wgpu::CommandEncoder& cmd);
@@ -358,7 +362,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
// Encode a sealed frame. Never touches the producer-visible recording state,
// so this may run concurrently with the producer's FIFO drains.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true);
// `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every
// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it
// after the last pass whose EFB copy the eye replays sample.
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true,
int32_t nativeRenderLastPass = INT32_MAX);
// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1
// when no pass does: everything after it exists only for the presented image.
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept;
// Replays only main-EFB passes into one Aurora-owned eye target. Native
// offscreen/EFB-copy passes are consumed from the mono render and are not
@@ -414,6 +425,39 @@ bool is_offscreen() noexcept;
uint32_t get_sample_count() noexcept;
void clear_caches() noexcept;
// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery,
// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the
// frame's queries are resolved into a small ring of readback buffers and the completed frames'
// durations are summed per category until gpu_timing_report() consumes them. Off by default:
// aurora.cpp enables it together with the Android frame-rate log.
enum class GpuTimingCategory : uint8_t {
Mono, // the native (desktop) render of the recorded GX passes
EyeLeft, // stereo replay of the left eye
EyeRight, // stereo replay of the right eye
Interpolated, // interpolated presentation slots
VirtualScreen, // the 2D virtual screen built for each eye
Panel, // the in-headset settings panel
EfbCopy, // EFB copy format conversions
Palette, // palette (TLUT) texture conversions
DepthPeek, // the depth snapshot compute pass
Snapshot, // presentation snapshot and its ImGui pass
Present, // the desktop presentation copy
Count,
};
void gpu_timing_set_enabled(bool enabled) noexcept;
bool gpu_timing_enabled() noexcept;
// Opens the current frame's query slot; a frame whose slot is still being read back is skipped.
void gpu_timing_begin_frame() noexcept;
// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted.
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept;
// Resolves the open frame's queries on `encoder`, which must be the frame's last submission.
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept;
// After that submission: starts the readback of the resolved queries.
void gpu_timing_after_submit() noexcept;
// Per-frame averages of the frames read back since the last call, formatted for the log, or
// an empty string when nothing was measured.
std::string gpu_timing_report();
namespace tex_palette_conv {
struct ConvRequest;
} // namespace tex_palette_conv
+2
View File
@@ -1,4 +1,5 @@
#include "depth_peek.hpp"
#include "common.hpp"
#include "../dolphin/vi/vi_internal.hpp"
#include "../gx/gx.hpp"
@@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV
const wgpu::ComputePassDescriptor passDescriptor{
.label = "Depth Peek Compute Pass",
.timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek),
};
const auto pass = cmd.BeginComputePass(&passDescriptor);
pass.SetPipeline(g_pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_copy_conv.hpp"
#include "common.hpp"
#include "tex_copy_format_contract.hpp"
#include "../internal.hpp"
@@ -651,6 +652,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
.label = "TexCopyConv Pass",
.colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy),
};
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline);
+2
View File
@@ -1,4 +1,5 @@
#include "tex_palette_conv.hpp"
#include "common.hpp"
#include "../internal.hpp"
#include "../webgpu/gpu.hpp"
@@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
.label = "TexPaletteConv Pass",
.colorAttachmentCount = colorAttachments.size(),
.colorAttachments = colorAttachments.data(),
.timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette),
};
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
pass.SetPipeline(pipeline);
+84 -43
View File
@@ -5,6 +5,7 @@
#include "../gfx/texture_replacement.hpp"
#include "dolphin/gx/GXAurora.h"
#include "gx.hpp"
#include "native_wheel.hpp"
#include "gx_fmt.hpp"
#include "pipeline.hpp"
#include "shader_info.hpp"
@@ -1929,7 +1930,8 @@ static u32 calculate_last_vtx_size(GXVtxFmt fmt) {
static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange,
uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature,
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride);
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride,
NativeWheelArray* nativeWheel);
// The per-draw geometry signature, matrix-usage mask and draw-identity hashes exist purely to feed frame interpolation
// (build_uniform consumes them only after its `frame_interpolation_fps() == 0` early-out).
@@ -1959,6 +1961,25 @@ static uint32_t matrix_index_prefix_size(GXVtxFmt fmt) noexcept {
return size;
}
// Which animated vertex array, if any, this draw takes (native_wheel.hpp). Called once per draw, before the merge
// test, because a merged draw renders through the binding the draw it folds into resolved.
static NativeWheelArray* resolve_native_wheel(GXVtxFmt fmt, const uint8_t* vertices, u16 vtxCount,
uint32_t vtxStride) noexcept {
// A direct-position draw reads no array at all, and g_gxState.arrays[GX_VA_POS] then still holds whatever was
// bound last, which must not be matched against.
if (g_gxState.vtxDesc[GX_VA_POS] != GX_INDEX8 && g_gxState.vtxDesc[GX_VA_POS] != GX_INDEX16)
LIKELY { return nullptr; }
const auto& array = g_gxState.arrays[GX_VA_POS];
if (nativeWheelArrays.empty())
LIKELY {
if (!nativeWheelPreviousSources.empty())
UNLIKELY { native_wheel_note_outside(array.data); }
return nullptr;
}
return native_wheel_array(array, vertices, static_cast<u32>(vtxCount) * vtxStride, vtxStride,
matrix_index_prefix_size(fmt));
}
// Screen-space bounds of a simple orthographic rectangle or line, textured or
// not: MKW's split-screen partition is a layout picture pane (a one-pixel quad
// sampling a pattern texture), so texture use cannot disqualify a candidate.
@@ -2185,6 +2206,10 @@ static ArrayRef<u16> offset_index_template(const CachedIndexTemplate& indexTempl
struct CachedPipelineState {
gfx::PipelineRef ref = 0;
const PipelineConfig* config = nullptr;
mutable gfx::PipelineRef stereoRef = 0;
mutable gfx::PipelineRef screenRef = 0;
mutable gfx::PipelineRef stereoScreenRef = 0;
HashType configHash = 0;
// Carried here so the draw can be recorded without keeping the PipelineConfig that produced it alive; it is the only
// field of the config the draw itself still needs.
@@ -2210,6 +2235,7 @@ static const CachedPipelineState& cached_pipeline_state(const PipelineConfig& co
entry.config = config;
entry.state = {
.ref = gfx::pipeline_ref(config),
.config = &entry.config,
.configHash = hash,
.dstAlpha = config.dstAlpha,
.shaderInfo = build_shader_info(config.shaderConfig),
@@ -2252,37 +2278,20 @@ static const CachedPipelineState& resolve_pipeline_state(GXPrimitive prim, GXVtx
return state;
}
// Exact Screen Depth changes shader outputs but no GX pipeline state. Resolve a
// sibling pipeline only for orthographic draws that can actually reach the VR
// screen, and memoize it with the same state epoch as the ordinary pipeline.
static gfx::PipelineRef resolve_exact_screen_depth_pipeline(GXPrimitive prim, GXVtxFmt fmt) {
struct Memo {
gfx::PipelineRef ref = 0;
u32 generation = 0;
u32 sampleCount = 0;
GXPrimitive prim = static_cast<GXPrimitive>(0);
GXVtxFmt fmt = static_cast<GXVtxFmt>(0);
};
static Memo memo{};
const u32 sampleCount = gfx::get_sample_count();
const u32 generation = g_gxState.pipelineStateGeneration;
if (memo.ref != 0 && memo.generation == generation && memo.sampleCount == sampleCount && memo.prim == prim &&
memo.fmt == fmt)
LIKELY { return memo.ref; }
PipelineConfig config{};
populate_pipeline_config(config, prim, fmt);
config.shaderConfig.exactScreenDepth = 1;
const gfx::PipelineRef ref = gfx::pipeline_ref(config);
memo = Memo{
.ref = ref,
.generation = generation,
.sampleCount = sampleCount,
.prim = prim,
.fmt = fmt,
};
return ref;
// Lazily cache eye-format siblings alongside the ordinary pipeline. Steady-state
// draws only read the refs: no extra config population/hashing on the Quest CPU.
// Shader modules are shared by the depth-format variants.
static void resolve_replay_pipelines(const CachedPipelineState& state, bool screen) {
if (state.stereoRef && (!screen || state.stereoScreenRef)) return;
PipelineConfig config = *state.config;
config.stereoStencil = 1;
if (!state.stereoRef) state.stereoRef = gfx::pipeline_ref(config);
if (screen && !state.stereoScreenRef) {
config.shaderConfig.exactScreenDepth = 1;
state.stereoScreenRef = gfx::pipeline_ref(config);
config.stereoStencil = 0;
state.screenRef = gfx::pipeline_ref(config);
}
}
bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, uint16_t vtxCount, uint32_t vertexBytes) {
@@ -2321,7 +2330,8 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui
const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{};
handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature,
interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0,
interpolationIdentityActive, vertices, vtxSize);
interpolationIdentityActive, vertices, vtxSize,
resolve_native_wheel(fmt, vertices, vtxCount, vtxSize));
return true;
}
@@ -2354,13 +2364,21 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
gfx::Range vertRange = push_draw_vertices(vertices, vtxCount, vtxSize);
pos += totalVtxBytes;
// Try to merge with previous draw call
// The animated vertex array this draw takes is decided per draw, and the decision is part of what a merge would
// share, so resolve it here and hand the result to handle_draw_unmerged rather than deciding twice.
NativeWheelArray* const nativeWheel = resolve_native_wheel(fmt, vertices, vtxCount, vtxSize);
// Try to merge with previous draw call.
if (!g_gxState.stateDirty && !(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC))
LIKELY {
auto* lastDraw = gfx::get_last_draw_command<DrawData>();
// Only if the previous draw call was a single instance draw (no lines/points handling)
// Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw
// that resolved the same animated array: the merged whole renders through that draw's binding. Anything the
// decision cache cannot vouch for (a command it was not recorded against) stays unmerged.
if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS &&
lastDraw->instanceCount == 1)
lastDraw->instanceCount == 1 &&
(nativeWheelArrays.empty() ||
(nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel)))
LIKELY {
const auto& indexTemplate = cached_index_template(prim, vtxCount);
const auto indices = offset_index_template(indexTemplate, lastDraw->vtxCount);
@@ -2389,13 +2407,14 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{};
handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature,
interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0,
interpolationIdentityActive, vertices, vtxSize);
interpolationIdentityActive, vertices, vtxSize, nativeWheel);
return true;
}
static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange,
uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature,
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride) {
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride,
NativeWheelArray* nativeWheel) {
ZoneScoped;
// GX_CULL_ALL rasterizes nothing on hardware - no color, no depth.
if (g_gxState.cullMode == GX_CULL_ALL && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS)
@@ -2419,6 +2438,24 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
}
auto& array = g_gxState.arrays[i];
const u32 uploadStride = padded_upload_stride(array.stride);
if (i == GX_VA_POS && nativeWheel != nullptr)
UNLIKELY {
static unsigned nativeWheelDrawLogs = 0;
if (nativeWheelDrawLogs++ < 4) Log.info("Native steering wheel: animated local vehicle vertex array");
// Never populate the shared source's cache with the animated copy: later draws of the same asset must
// still see the original vertices. The copy takes the same padded upload path as the original.
if (nativeWheel->uploaded.size == 0 || nativeWheel->uploadedStride != uploadStride) {
AttrArray animated{};
animated.data = nativeWheel->bytes.data();
animated.size = array.size;
animated.stride = array.stride;
animated.le = array.le;
nativeWheel->uploaded = push_vertex_array(animated, uploadStride);
nativeWheel->uploadedStride = uploadStride;
}
ranges.vaRanges[0] = nativeWheel->uploaded;
continue;
}
if (array.cachedRange.size > 0 && array.cachedStride == uploadStride) {
ranges.vaRanges[i - GX_VA_POS] = array.cachedRange;
} else {
@@ -2459,10 +2496,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
const bool perspective = g_gxState.projType == GX_PERSPECTIVE;
const auto uniformRanges = build_uniform(info, vertRange.offset, ranges, drawIdentity, perspective, usedPnMtxMask);
const auto& replayLayout = uniformRanges.replayLayout;
const gfx::PipelineRef exactScreenDepthPipeline =
aurora::stereo_frame_provider_active() && !replayLayout.perspective && !replayLayout.nativeEfbEffect
? resolve_exact_screen_depth_pipeline(prim, fmt)
: 0;
const bool stereo = aurora::stereo_frame_provider_active();
const bool screen = !replayLayout.perspective && !replayLayout.nativeEfbEffect;
if (stereo) resolve_replay_pipelines(pipelineState, screen);
s_lastDrawRecordedInterpolation = interpolationIdentityActive;
uint32_t instanceCount = 1;
@@ -2475,7 +2511,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
}
gfx::push_draw_command(DrawData{
.pipeline = pipeline,
.exactScreenDepthPipeline = exactScreenDepthPipeline,
.exactScreenDepthPipeline = stereo && screen ? pipelineState.screenRef : 0,
.stereoPipeline = stereo ? pipelineState.stereoRef : 0,
.stereoScreenPipeline = stereo && screen ? pipelineState.stereoScreenRef : 0,
.vertRange = vertRange,
.idxRange = idxRange,
.uniformRange = uniformRanges.current,
@@ -2490,6 +2528,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
.dstAlpha = pipelineState.dstAlpha,
.screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride),
});
// What the next draw must match to be allowed to fold into this one.
nativeWheelLastDrawCommand = gfx::get_last_draw_command<DrawData>();
nativeWheelLastDecision = nativeWheel;
g_gxState.stateDirty = false;
}
+9 -1
View File
@@ -1609,10 +1609,18 @@ static inline wgpu::PrimitiveState to_primitive_state(GXCullMode gx_cullMode) {
wgpu::RenderPipeline build_pipeline(const PipelineConfig& config, ArrayRef<wgpu::VertexBufferLayout> vtxBuffers,
wgpu::ShaderModule shader, const char* label) noexcept {
ZoneScoped;
const bool maskCockpit = config.stereoStencil && config.shaderConfig.exactScreenDepth;
const wgpu::StencilFaceState stencil{
.compare = maskCockpit ? wgpu::CompareFunction::Equal : wgpu::CompareFunction::Always,
};
const wgpu::DepthStencilState depthStencil{
.format = g_graphicsConfig.depthFormat,
.format = config.stereoStencil ? wgpu::TextureFormat::Depth24PlusStencil8 : g_graphicsConfig.depthFormat,
.depthWriteEnabled = config.depthUpdate,
.depthCompare = config.depthCompare ? to_compare_function(config.depthFunc) : wgpu::CompareFunction::Always,
.stencilFront = stencil,
.stencilBack = stencil,
.stencilReadMask = 1,
.stencilWriteMask = 0,
};
const auto blendState = to_blend_state(config.blendMode, config.blendFacSrc, config.blendFacDst, config.blendOp,
config.pixelFmt, config.dstAlpha);
+119
View File
@@ -0,0 +1,119 @@
// SPDX-License-Identifier: GPL-3.0-or-later
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
//
// The VR first-person camera animates the local vehicle's steering wheel by
// handing Aurora a rotated copy of one of the vehicle's position arrays. The
// copy applies only to a draw that binds that exact array *and* carries the
// local vehicle's model-view matrix, so an opponent sharing the asset keeps
// the original vertices. Everything here runs on the GX (command processor)
// thread; the set/clear entry points are posted there by the runtime.
#pragma once
#include "gx.hpp"
#include <algorithm>
#include <vector>
#include <cstring>
#include <cmath>
#include <atomic>
#include <aurora/native_wheel_match.hpp>
namespace aurora::gx {
struct NativeWheelArray {
const void* source{};
std::vector<uint8_t> bytes;
std::array<float,12> modelView{};
gfx::Range uploaded{};
// Element stride of `uploaded`, which differs from the array's when the upload is padded.
uint32_t uploadedStride=0;
};
inline std::vector<NativeWheelArray> nativeWheelArrays;
inline uint32_t nativeWheelMatches=0;
inline std::atomic<uint32_t> nativeWheelLastMatches{0};
// A draw folds into the previous one by appending its vertices to that draw's
// range, so the merged whole renders through the *first* draw's array binding:
// two draws may only merge when they resolved the same replacement. The
// decision below is therefore taken once per draw (the ownership walk is far
// too costly to repeat) and kept with the command it was recorded for. It is
// cleared with the set, which the runtime posts once a frame, so a command
// address a later frame's list reuses can never be read as a hit.
inline NativeWheelArray* nativeWheelLastDecision=nullptr;
inline const void* nativeWheelLastDrawCommand=nullptr;
// For the host log: why draws of the replaced arrays did or did not take them.
struct NativeWheelDiagnostics {
uint32_t sets=0; // replacement sets cleared since the last report
uint32_t boundDraws=0; // draws that bound a replacement's source array
uint32_t oversizeDraws=0; // ... whose bound range was larger than the replacement
uint32_t outsideDraws=0; // draws that bound a cleared set's source while no set was active
uint32_t matchedDraws=0;
float bestError=INFINITY; // smallest largest-element difference of a position matrix
bool bestIndexed=false;
std::array<float,12> bestMatrix{};
std::array<float,12> expected{};
};
inline NativeWheelDiagnostics nativeWheelDiagnostics;
inline std::vector<const void*> nativeWheelPreviousSources;
inline uint32_t nativeWheelClears=0;
inline uint32_t nativeWheelReports=0;
inline void native_wheel_note_outside(const void* source) {
for(const void* previous:nativeWheelPreviousSources)
if(previous==source) { ++nativeWheelDiagnostics.outsideDraws;return; }
}
inline void native_wheel_note_bound(const NativeWheelArray& replacement,bool indexedMatrix) {
auto& diagnostics=nativeWheelDiagnostics;
++diagnostics.boundDraws;
for(uint32_t slot=0;slot<MaxPnMtx;++slot) {
if(!indexedMatrix && slot!=g_gxState.currentPnMtx) continue;
const auto* matrix=reinterpret_cast<const float*>(&g_gxState.pnMtx[slot].pos);
float error=0;
for(int i=0;i<12;++i) error=std::max(error,std::abs(matrix[i]-replacement.modelView[i]));
if(error<diagnostics.bestError) {
diagnostics.bestError=error;
diagnostics.bestIndexed=indexedMatrix;
std::memcpy(diagnostics.bestMatrix.data(),matrix,sizeof(float)*12);
diagnostics.expected=replacement.modelView;
}
}
}
// Called as a set is cleared (GXAurora.cpp): one host log line at about half a
// second, five seconds and a minute of sets.
void native_wheel_report();
inline bool native_wheel_source(const void* source) {
for(const auto& replacement:nativeWheelArrays) if(source==replacement.source) return true;
return false;
}
inline NativeWheelArray* native_wheel_array(const AttrArray& array,const uint8_t* vertices,
uint32_t vertexBytes,uint32_t vertexStride,uint32_t positionOffset) {
const bool indexedMatrix=g_gxState.vtxDesc[GX_VA_PNMTXIDX]==GX_DIRECT;
if(!indexedMatrix && g_gxState.currentPnMtx>=MaxPnMtx) return nullptr;
for(auto& replacement:nativeWheelArrays) {
if(array.data!=replacement.source) continue;
if(array.size>replacement.bytes.size()) { ++nativeWheelDiagnostics.oversizeDraws;continue; }
native_wheel_note_bound(replacement,indexedMatrix);
// A matching asset alone would also animate an opponent. Multi-joint
// models need a per-position ownership check, not a blanket exclusion.
uint16_t matching=0;
for(uint32_t slot=0;slot<MaxPnMtx;++slot) {
if(!indexedMatrix && slot!=g_gxState.currentPnMtx) continue;
const auto* matrix=reinterpret_cast<const float*>(&g_gxState.pnMtx[slot].pos);
bool matches=true;
for(int i=0;i<12;++i) {
uint32_t bits;std::memcpy(&bits,&matrix[i],4);
if((bits&0x7f800000u)==0x7f800000u ||
std::abs(matrix[i]-replacement.modelView[i])>(i%4==3?0.1f:0.002f)) { matches=false;break; }
}
if(matches) matching|=uint16_t(1u<<slot);
}
if(!matching) continue;
if(indexedMatrix && !NativeWheelDrawMatches(
{static_cast<const uint8_t*>(array.data),array.size},
{replacement.bytes.data(),array.size},array.stride,
{vertices,vertexBytes},vertexStride,positionOffset,
g_gxState.vtxDesc[GX_VA_POS]==GX_INDEX8?1:2,matching)) continue;
++nativeWheelMatches;++nativeWheelDiagnostics.matchedDraws;return &replacement;
}
return nullptr;
}
}
+6 -2
View File
@@ -11,6 +11,9 @@ struct DrawData {
// Same GX state with exact fragment-depth export enabled. Bound only when
// this draw is actually reprojected onto the VR virtual screen.
gfx::PipelineRef exactScreenDepthPipeline;
// Eye depth/stencil format siblings. Shader modules are shared with mono.
gfx::PipelineRef stereoPipeline = 0;
gfx::PipelineRef stereoScreenPipeline = 0;
gfx::Range vertRange;
gfx::Range idxRange;
gfx::Range uniformRange;
@@ -29,7 +32,7 @@ struct DrawData {
std::optional<gfx::stereo_replay::SubviewRect> screenRect;
};
constexpr uint32_t GXPipelineConfigVersion = 20;
constexpr uint32_t GXPipelineConfigVersion = 21;
constexpr GXFogType effective_pipeline_fog_type(GXFogType fogType, GXZTexOp zTextureOp, bool zCompLocBeforeTex,
GXBlendMode blendMode, GXLogicOp logicOp) noexcept {
@@ -41,6 +44,7 @@ constexpr GXFogType effective_pipeline_fog_type(GXFogType fogType, GXZTexOp zTex
struct PipelineConfig {
uint32_t version = GXPipelineConfigVersion;
uint32_t msaaSamples = 1;
uint32_t stereoStencil = 0;
ShaderConfig shaderConfig;
GXCompare depthFunc;
GXCullMode cullMode;
@@ -61,7 +65,7 @@ inline bool valid_pipeline_config(const PipelineConfig& config) noexcept {
};
const bool validSamples =
config.msaaSamples == 1 || config.msaaSamples == 2 || config.msaaSamples == 4 || config.msaaSamples == 8;
return config.version == GXPipelineConfigVersion && validSamples && in_range(config.depthFunc, GX_ALWAYS) &&
return config.version == GXPipelineConfigVersion && validSamples && config.stereoStencil <= 1 && in_range(config.depthFunc, GX_ALWAYS) &&
in_range(config.cullMode, GX_CULL_ALL) && in_range(config.blendMode, GX_BM_SUBTRACT) &&
in_range(config.blendFacSrc, GX_BL_INVDSTALPHA) && in_range(config.blendFacDst, GX_BL_INVDSTALPHA) &&
in_range(config.blendOp, GX_LO_SET) && in_range(config.pixelFmt, GX_PF_YUV420);
+79 -2
View File
@@ -13,6 +13,7 @@
#include "fs_helper.hpp"
#include "internal.hpp"
#include "stereo_overlay.hpp"
#include "webgpu/gpu.hpp"
#include "window.hpp"
@@ -28,7 +29,11 @@ static std::string g_imguiLog{};
static bool g_useSdlRenderer = false;
// Set once ImGui::Render() has produced this frame's draw data. Interpolation encodes up to four
// ImGui passes per frame, and every one of them used to rebuild the draw lists from scratch.
static bool g_frameDataBuilt = false;
static bool g_frameDataBuilt = true;
// Host-owned frames (see imgui.hpp). Once the host begins one, aurora never calls new_frame() or
// ImGui::Render() itself; the sealed frame carries the host's copy of the draw data instead.
static bool g_hostFrames = false;
static bool g_hostFrameOpen = false;
static std::vector<SDL_Texture*> g_sdlTextures;
static std::vector<wgpu::Texture> g_wgpuTextures;
@@ -179,7 +184,7 @@ void new_frame(const AuroraWindowSize& size) noexcept {
void render_frame_data() noexcept {
ZoneScoped;
if (g_frameDataBuilt) {
if (g_frameDataBuilt || g_hostFrames) {
return;
}
ImGui::Render();
@@ -190,6 +195,11 @@ void render_frame_data() noexcept {
void render(const wgpu::RenderPassEncoder& pass) noexcept {
ZoneScoped;
if (g_hostFrames) {
// The shared context's draw data belongs to the host's current frame now;
// a sealed frame without a host copy has nothing safe to draw.
return;
}
render_frame_data();
auto* data = ImGui::GetDrawData();
@@ -205,6 +215,71 @@ void render(const wgpu::RenderPassEncoder& pass) noexcept {
}
}
struct HostFrame {
ImDrawData data{};
std::vector<ImDrawList*> lists;
~HostFrame() {
for (ImDrawList* list : lists) {
IM_DELETE(list);
}
}
};
void host_frame_begin(const AuroraWindowSize& size) noexcept {
g_hostFrames = true;
if (g_hostFrameOpen) {
return;
}
if (!g_frameDataBuilt) {
// aurora started this frame itself before the host took over: adopt it.
g_hostFrameOpen = true;
return;
}
new_frame(size);
g_hostFrameOpen = true;
}
HostFramePtr host_frame_end() noexcept {
ZoneScoped;
if (!g_hostFrameOpen) {
host_frame_begin(window::get_window_size());
}
ImGui::Render();
ImDrawData* source = ImGui::GetDrawData();
source->FramebufferScale = ImGui::GetIO().DisplayFramebufferScale;
auto frame = std::make_shared<HostFrame>();
frame->data = *source;
frame->data.CmdLists.clear();
frame->lists.reserve(static_cast<size_t>(source->CmdListsCount));
for (int i = 0; i < source->CmdListsCount; ++i) {
const ImDrawList* src = source->CmdLists[i];
ImDrawList* copy = IM_NEW(ImDrawList)(src->_Data);
copy->CmdBuffer = src->CmdBuffer;
copy->IdxBuffer = src->IdxBuffer;
copy->VtxBuffer = src->VtxBuffer;
copy->Flags = src->Flags;
frame->lists.push_back(copy);
frame->data.CmdLists.push_back(copy);
}
g_hostFrameOpen = false;
g_frameDataBuilt = true;
return frame;
}
bool host_frames_active() noexcept { return g_hostFrames; }
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept { return &frame.data; }
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept {
ZoneScoped;
if (g_useSdlRenderer || data == nullptr) {
return;
}
pass.PushDebugGroup("Aurora: Dear Imgui");
ImGui_ImplWGPU_RenderDrawData(const_cast<ImDrawData*>(data), pass.Get());
pass.PopDebugGroup();
}
StereoOverlay latch_stereo_overlay() noexcept {
std::lock_guard lock(g_stereoOverlayMutex);
return g_stereoOverlay;
@@ -274,6 +349,8 @@ ImTextureID aurora_imgui_add_texture(uint32_t width, uint32_t height, const void
return aurora::imgui::add_texture(width, height, static_cast<const uint8_t*>(rgba8));
}
void aurora_set_stereo_panel_layer(bool enabled) { aurora::stereo_overlay::set_layer_mode(enabled); }
void aurora_imgui_set_stereo_overlay(ImDrawData* drawData, float widthFraction) {
std::lock_guard lock(aurora::imgui::g_stereoOverlayMutex);
aurora::imgui::g_stereoOverlay = {
+14
View File
@@ -1,6 +1,7 @@
#pragma once
#include <aurora/event.h>
#include <memory>
union SDL_Event;
struct ImDrawData;
@@ -32,4 +33,17 @@ StereoOverlay latch_stereo_overlay() noexcept;
// uniform for every pass, so a pass whose display size differs from the desktop's must be submitted
// before the next pass is recorded.
bool render_draw_data(const wgpu::RenderPassEncoder& pass, ImDrawData* data) noexcept;
// Host-owned ImGui frames. The host starts each frame on its own thread with host_frame_begin() and
// closes it with host_frame_end(), which renders the frame and copies its draw data out of the shared
// context. The copy is what the sealed frame replays, so the host may start the next frame while the
// worker still encodes this one, and aurora stops starting frames itself once the host has begun one.
struct HostFrame;
using HostFramePtr = std::shared_ptr<HostFrame>;
void host_frame_begin(const AuroraWindowSize& size) noexcept;
HostFramePtr host_frame_end() noexcept;
bool host_frames_active() noexcept;
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept;
// Renders a host frame's copied draw data in place of the shared context's.
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept;
} // namespace aurora::imgui
+55 -2
View File
@@ -11,6 +11,7 @@
#include "tracy/Tracy.hpp"
#include <array>
#include <atomic>
#include <cmath>
namespace aurora::stereo_overlay {
@@ -74,8 +75,12 @@ struct State {
std::array<wgpu::BindGroup, AURORA_STEREO_EYE_COUNT> bindGroups;
float widthFraction = 0.f;
bool visible = false;
// Stands in for the panel in a layer while it is not showing.
webgpu::TextureWithSampler transparent;
bool transparentCleared = false;
};
State g_state;
std::atomic_bool g_layerMode{false};
bool ensure_pipeline() {
auto& state = g_state;
@@ -239,6 +244,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar
.label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
};
const auto pass = encoder.BeginRenderPass(&descriptor);
pass.SetPipeline(state.pipeline);
@@ -294,6 +300,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
.label = "Headset panel ImGui pass",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
};
bool drawn = false;
{
@@ -312,7 +319,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
void composite_immersive(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& eye,
const Mat4x4<float>& eyeFrustum, const Mat3x4<float>& viewFromCenter,
uint32_t eyeIndex) noexcept {
if (!g_state.visible) {
if (!g_state.visible || layer_mode()) {
return;
}
float screenWidth = 0.f;
@@ -329,7 +336,7 @@ void composite_immersive(const wgpu::CommandEncoder& encoder, const wgpu::Textur
void composite_flat(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& eye, const wgpu::Extent3D& size,
uint32_t eyeIndex) noexcept {
if (!g_state.visible || size.width == 0 || size.height == 0) {
if (!g_state.visible || layer_mode() || size.width == 0 || size.height == 0) {
return;
}
const float imageAspect = static_cast<float>(size.width) / static_cast<float>(size.height);
@@ -338,6 +345,52 @@ void composite_flat(const wgpu::CommandEncoder& encoder, const wgpu::TextureView
eyeIndex);
}
void set_layer_mode(bool enabled) noexcept { g_layerMode.store(enabled, std::memory_order_release); }
bool layer_mode() noexcept { return g_layerMode.load(std::memory_order_acquire); }
bool layer_source(const wgpu::CommandEncoder& encoder, uint32_t width, uint32_t height, stereo::EyeImage& out) noexcept {
auto& state = g_state;
const auto format = webgpu::g_graphicsConfig.surfaceConfiguration.format;
if (width == 0 || height == 0) {
return false;
}
if (state.visible && state.panel.texture && state.panel.size.width == width && state.panel.size.height == height &&
state.panel.format == format) {
out = {.texture = &state.panel.texture, .view = &state.panel.view, .size = state.panel.size, .format = format};
return true;
}
if (!state.transparent.texture || state.transparent.size.width != width ||
state.transparent.size.height != height || state.transparent.format != format) {
state.transparent = webgpu::create_render_texture(width, height, false);
state.transparentCleared = false;
if (state.transparent.size.width != width || state.transparent.size.height != height) {
state.transparent = {};
return false;
}
}
if (!state.transparentCleared) {
const std::array attachments{
wgpu::RenderPassColorAttachment{
.view = state.transparent.view,
.loadOp = wgpu::LoadOp::Clear,
.storeOp = wgpu::StoreOp::Store,
.clearValue = {.r = 0.0, .g = 0.0, .b = 0.0, .a = 0.0},
},
};
const wgpu::RenderPassDescriptor descriptor{
.label = "Headset panel layer clear",
.colorAttachmentCount = attachments.size(),
.colorAttachments = attachments.data(),
};
encoder.BeginRenderPass(&descriptor).End();
state.transparentCleared = true;
}
out = {.texture = &state.transparent.texture, .view = &state.transparent.view, .size = state.transparent.size,
.format = format};
return true;
}
void shutdown() noexcept { g_state = {}; }
} // namespace aurora::stereo_overlay
+16
View File
@@ -3,6 +3,8 @@
#include <aurora/math.hpp>
#include <webgpu/webgpu_cpp.h>
#include "stereo.hpp"
#include <cstdint>
struct ImDrawData;
@@ -28,6 +30,20 @@ void composite_immersive(const wgpu::CommandEncoder& encoder, const wgpu::Textur
void composite_flat(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& eye, const wgpu::Extent3D& size,
uint32_t eyeIndex) noexcept;
// Layer mode: the OpenXR backend shows the panel as its own compositor quad
// layer, sharp at any eye resolution, so it is no longer drawn into the eyes
// (both composite functions do nothing). Any thread; read by the frame worker.
void set_layer_mode(bool enabled) noexcept;
bool layer_mode() noexcept;
// Frame worker, inside a stereo sink: the image to copy into a panel layer of
// width x height in the eyes' format. That is the panel while it is showing at
// exactly that size, and otherwise a transparent image of that size, so a layer
// asked for before the panel's first frame (or after it closed) shows nothing.
// A transparent image is cleared once, by a pass recorded into `encoder`.
// False only when no image can be made.
bool layer_source(const wgpu::CommandEncoder& encoder, uint32_t width, uint32_t height, stereo::EyeImage& out) noexcept;
void shutdown() noexcept;
} // namespace aurora::stereo_overlay
+90 -38
View File
@@ -2,6 +2,7 @@
#include "../internal.hpp"
#include "../stereo.hpp"
#include "../stereo_overlay.hpp"
#include "gpu.hpp"
#if defined(_WIN32) && defined(WEBGPU_DAWN) && defined(DAWN_ENABLE_BACKEND_D3D12)
@@ -126,6 +127,11 @@ struct SharedFenceDxgiHandleWire {
void* handle = nullptr;
};
// The eyes, then the settings panel's layer image in a slot of its own so the
// eye intermediates are never resized for it.
constexpr uint32_t kPanelIndex = AURORA_D3D12_STEREO_MAX_TARGETS;
constexpr uint32_t kMaxImages = AURORA_D3D12_STEREO_MAX_TARGETS + 1;
struct IntermediateEye {
ComPtr<ID3D12Resource> resource;
wgpu::SharedTextureMemory memory;
@@ -153,8 +159,8 @@ struct InFlightCommand {
// both sides of every copy alive until this submission's fence completes;
// an eye-size change may otherwise replace the bridge intermediate while
// the GPU is still reading it.
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> sources;
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> destinations;
std::array<ComPtr<ID3D12Resource>, kMaxImages> sources;
std::array<ComPtr<ID3D12Resource>, kMaxImages> destinations;
};
class StereoBridge final {
@@ -209,38 +215,51 @@ public:
return WaitForGpuLocked();
}
bool SetTargets(uint64_t token, const AuroraD3D12StereoTarget* targets,
uint32_t targetCount) noexcept {
bool SetTargets(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t targetCount,
const AuroraD3D12StereoTarget* panel) noexcept {
if (token == 0 || targets == nullptr || targetCount == 0 ||
targetCount > AURORA_D3D12_STEREO_MAX_TARGETS) {
return false;
}
const auto valid = [](const AuroraD3D12StereoTarget& target) {
if (target.resource == nullptr || target.width == 0 || target.height == 0 ||
target.dxgiFormat == DXGI_FORMAT_UNKNOWN) {
return false;
}
const D3D12_RESOURCE_DESC desc = static_cast<ID3D12Resource*>(target.resource)->GetDesc();
return desc.Dimension == D3D12_RESOURCE_DIMENSION_TEXTURE2D && desc.Width >= target.width &&
desc.Height >= target.height && desc.DepthOrArraySize == 1 && desc.MipLevels == 1 &&
desc.SampleDesc.Count == 1 &&
same_copy_family(desc.Format, static_cast<DXGI_FORMAT>(target.dxgiFormat));
};
std::lock_guard lock(m_mutex);
if (m_framePending || m_encoded) {
return false;
}
for (uint32_t eye = 0; eye < targetCount; ++eye) {
if (targets[eye].resource == nullptr || targets[eye].width == 0 ||
targets[eye].height == 0 || targets[eye].dxgiFormat == DXGI_FORMAT_UNKNOWN) {
if (!valid(targets[eye])) {
return false;
}
auto* resource = static_cast<ID3D12Resource*>(targets[eye].resource);
const D3D12_RESOURCE_DESC desc = resource->GetDesc();
if (desc.Dimension != D3D12_RESOURCE_DIMENSION_TEXTURE2D ||
desc.Width < targets[eye].width || desc.Height < targets[eye].height ||
desc.DepthOrArraySize != 1 || desc.MipLevels != 1 || desc.SampleDesc.Count != 1 ||
!same_copy_family(desc.Format, static_cast<DXGI_FORMAT>(targets[eye].dxgiFormat))) {
return false;
}
m_targets[eye] = {
.resource = resource,
.width = targets[eye].width,
.height = targets[eye].height,
.format = static_cast<DXGI_FORMAT>(targets[eye].dxgiFormat),
};
}
for (uint32_t eye = targetCount; eye < m_targets.size(); ++eye) {
m_targets[eye] = {};
if (panel != nullptr && !valid(*panel)) {
return false;
}
m_targets = {};
m_imageCount = 0;
const auto add = [&](uint32_t index, const AuroraD3D12StereoTarget& target) {
m_targets[index] = {
.resource = static_cast<ID3D12Resource*>(target.resource),
.width = target.width,
.height = target.height,
.format = static_cast<DXGI_FORMAT>(target.dxgiFormat),
};
m_images[m_imageCount++] = index;
};
for (uint32_t eye = 0; eye < targetCount; ++eye) {
add(eye, targets[eye]);
}
if (panel != nullptr) {
add(kPanelIndex, *panel);
}
m_frameToken = token;
m_targetCount = targetCount;
@@ -307,7 +326,7 @@ private:
if (source.texture == nullptr || sourceFormat == DXGI_FORMAT_UNKNOWN ||
source.size.width != m_targets[eye].width || source.size.height != m_targets[eye].height ||
!same_copy_family(sourceFormat, m_targets[eye].format)) {
Log.error("Stereo eye {} does not match its OpenXR D3D12 target", eye);
Log.error("Stereo image {} does not match its OpenXR D3D12 target", eye);
return false;
}
if (intermediate.texture && intermediate.width == source.size.width &&
@@ -350,7 +369,9 @@ private:
wire.resource = intermediate.resource;
const wgpu::SharedTextureMemoryDescriptor memoryDescriptor{
.nextInChain = &wire.chain,
.label = eye == 0 ? "OpenXR left eye intermediate" : "OpenXR right eye intermediate",
.label = eye == 0 ? "OpenXR left eye intermediate"
: eye == 1 ? "OpenXR right eye intermediate"
: "OpenXR panel intermediate",
};
intermediate.memory = webgpu::g_device.ImportSharedTextureMemory(&memoryDescriptor);
if (!intermediate.memory) {
@@ -368,7 +389,9 @@ private:
return false;
}
const wgpu::TextureDescriptor textureDescriptor{
.label = eye == 0 ? "OpenXR left eye shared texture" : "OpenXR right eye shared texture",
.label = eye == 0 ? "OpenXR left eye shared texture"
: eye == 1 ? "OpenXR right eye shared texture"
: "OpenXR panel shared texture",
.usage = wgpu::TextureUsage::CopyDst,
.dimension = wgpu::TextureDimension::e2D,
.size = {source.size.width, source.size.height, 1},
@@ -391,12 +414,22 @@ private:
bool EncodeLocked(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) noexcept {
CollectCompletedCommandsLocked();
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
if (!EnsureIntermediate(eye, frame.eyes[eye])) {
std::array<stereo::EyeImage, kMaxImages> sources{};
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
if (eye == kPanelIndex) {
if (!stereo_overlay::layer_source(encoder, m_targets[eye].width, m_targets[eye].height, sources[eye])) {
return false;
}
} else {
sources[eye] = frame.eyes[eye];
}
if (!EnsureIntermediate(eye, sources[eye])) {
return false;
}
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
auto& intermediate = m_intermediates[eye];
const std::array fences{m_webgpuFence};
const std::array values{m_lastExternalFenceValue};
@@ -410,7 +443,8 @@ private:
}
if (intermediate.memory.BeginAccess(intermediate.texture, &begin) != wgpu::Status::Success) {
Log.error("Dawn BeginAccess failed for stereo eye {}", eye);
for (uint32_t begunEye = 0; begunEye < eye; ++begunEye) {
for (uint32_t m = 0; m < n; ++m) {
const uint32_t begunEye = m_images[m];
wgpu::SharedTextureMemoryEndAccessState end{};
m_intermediates[begunEye].memory.EndAccess(m_intermediates[begunEye].texture, &end);
m_intermediates[begunEye].initialized = end.initialized;
@@ -424,10 +458,11 @@ private:
// to one. If a later BeginAccess fails, the rollback above can therefore
// end the earlier accesses without leaving an unsubmitted copy that uses
// a texture after its access interval.
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
const auto& intermediate = m_intermediates[eye];
const wgpu::TexelCopyTextureInfo source{
.texture = *frame.eyes[eye].texture,
.texture = *sources[eye].texture,
.mipLevel = 0,
.origin = {},
.aspect = wgpu::TextureAspect::All,
@@ -446,7 +481,8 @@ private:
bool EndAccessLocked() noexcept {
bool success = true;
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
auto& intermediate = m_intermediates[eye];
if (!intermediate.accessBegun) {
success = false;
@@ -467,8 +503,8 @@ private:
bool EnqueueNativeCopyLocked() noexcept {
ComPtr<ID3D12CommandAllocator> allocator;
ComPtr<ID3D12GraphicsCommandList> list;
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> sources;
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> destinations;
std::array<ComPtr<ID3D12Resource>, kMaxImages> sources;
std::array<ComPtr<ID3D12Resource>, kMaxImages> destinations;
if (FAILED(m_device->CreateCommandAllocator(D3D12_COMMAND_LIST_TYPE_DIRECT,
IID_PPV_ARGS(&allocator))) ||
FAILED(m_device->CreateCommandList(0, D3D12_COMMAND_LIST_TYPE_DIRECT, allocator.Get(),
@@ -477,7 +513,8 @@ private:
return false;
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
const auto& source = m_intermediates[eye];
const auto& destination = m_targets[eye];
sources[eye] = source.resource;
@@ -564,6 +601,7 @@ private:
}
m_frameToken = 0;
m_targetCount = 0;
m_imageCount = 0;
m_framePending = false;
m_encoded = false;
}
@@ -625,8 +663,11 @@ private:
ComPtr<ID3D12CommandQueue> m_queue;
ComPtr<ID3D12Fence> m_fence;
wgpu::SharedFence m_webgpuFence;
std::array<IntermediateEye, AURORA_D3D12_STEREO_MAX_TARGETS> m_intermediates{};
std::array<PendingTarget, AURORA_D3D12_STEREO_MAX_TARGETS> m_targets{};
std::array<IntermediateEye, kMaxImages> m_intermediates{};
std::array<PendingTarget, kMaxImages> m_targets{};
// The slots of m_targets this frame copies into, eyes first.
std::array<uint32_t, kMaxImages> m_images{};
uint32_t m_imageCount = 0;
std::vector<InFlightCommand> m_commands;
AuroraD3D12StereoSubmittedCallback m_callback = nullptr;
void* m_userdata = nullptr;
@@ -701,7 +742,13 @@ bool aurora_d3d12_set_stereo_targets(uint64_t frameToken,
const AuroraD3D12StereoTarget* targets,
uint32_t targetCount) {
using namespace aurora::d3d12_interop;
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount);
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, nullptr);
}
bool aurora_d3d12_set_stereo_targets_with_panel(uint64_t frameToken, const AuroraD3D12StereoTarget* targets,
uint32_t targetCount, const AuroraD3D12StereoTarget* panel) {
using namespace aurora::d3d12_interop;
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, panel);
}
bool aurora_d3d12_cancel_stereo_targets(uint64_t frameToken) {
@@ -744,6 +791,11 @@ bool aurora_d3d12_set_stereo_targets(uint64_t, const AuroraD3D12StereoTarget*, u
return false;
}
bool aurora_d3d12_set_stereo_targets_with_panel(uint64_t, const AuroraD3D12StereoTarget*, uint32_t,
const AuroraD3D12StereoTarget*) {
return false;
}
bool aurora_d3d12_cancel_stereo_targets(uint64_t) { return false; }
bool aurora_d3d12_disable_stereo_bridge() { return true; }
+12
View File
@@ -80,6 +80,7 @@ wgpu::Instance g_instance;
static wgpu::AdapterInfo g_adapterInfo;
static wgpu::SurfaceCapabilities g_surfaceCapabilities;
bool g_bcTexturesSupported;
bool g_timestampQueriesSupported = false;
// Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the
// callback free of logging, allocation, teardown and renderer state mutation.
static std::atomic_bool g_deviceLost{false};
@@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) {
g_bcTexturesSupported = true;
requiredFeatures.push_back(feature);
}
// Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature
// costs nothing until a pass carries timestamp writes.
if (feature == wgpu::FeatureName::TimestampQuery) {
g_timestampQueriesSupported = true;
requiredFeatures.push_back(feature);
}
// The presenter calls device and queue methods while the frame worker encodes, which Dawn only
// supports with this feature; without it the two race inside the device's dynamic uploader.
if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) {
@@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) {
if (g_backendType == wgpu::BackendType::Vulkan) {
enableToggles.push_back("vulkan_monolithic_pipeline_cache");
}
// Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants
// the raw values.
const std::array<const char*, 1> disableToggles{"timestamp_quantization"};
const wgpu::DawnTogglesDescriptor togglesDescriptor({
.nextInChain = &cacheDescriptor,
.enabledToggleCount = enableToggles.size(),
.enabledToggles = enableToggles.data(),
.disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0,
.disabledToggles = disableToggles.data(),
});
#endif
wgpu::DeviceDescriptor deviceDescriptor;
+2
View File
@@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline;
extern wgpu::BindGroup g_CopyBindGroup;
extern wgpu::Instance g_instance;
extern bool g_bcTexturesSupported;
// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*).
extern bool g_timestampQueriesSupported;
bool initialize(AuroraBackend backend);
void shutdown();
+97 -48
View File
@@ -2,6 +2,7 @@
#include "../internal.hpp"
#include "../stereo.hpp"
#include "../stereo_overlay.hpp"
#include "gpu.hpp"
#if defined(__ANDROID__) && defined(WEBGPU_DAWN)
@@ -105,6 +106,11 @@ struct Import {
bool accessBegun = false;
};
// The eyes, then the settings panel's layer image in a slot of its own.
constexpr uint32_t kPanelIndex = AURORA_VULKAN_STEREO_MAX_TARGETS;
constexpr uint32_t kMaxImages = AURORA_VULKAN_STEREO_MAX_RELEASES;
using Releases = std::array<AuroraVulkanStereoRelease, kMaxImages>;
struct PendingTarget {
AHardwareBuffer* buffer = nullptr;
uint32_t width = 0;
@@ -146,8 +152,8 @@ public:
return true;
}
bool SetTargets(uint64_t token, const AuroraVulkanStereoTarget* targets,
uint32_t targetCount) noexcept {
bool SetTargets(uint64_t token, const AuroraVulkanStereoTarget* targets, uint32_t targetCount,
const AuroraVulkanStereoTarget* panel) noexcept {
if (token == 0 || targets == nullptr || targetCount == 0 ||
targetCount > AURORA_VULKAN_STEREO_MAX_TARGETS) {
return false;
@@ -157,25 +163,36 @@ public:
return false;
}
const int64_t auroraFormat = to_vk_format(m_auroraFormat);
const auto valid = [&](const AuroraVulkanStereoTarget& target) {
return target.buffer != nullptr && target.width != 0 && target.height != 0 &&
same_copy_family(target.vkFormat, auroraFormat);
};
for (uint32_t eye = 0; eye < targetCount; ++eye) {
const auto& target = targets[eye];
if (target.buffer == nullptr || target.width == 0 || target.height == 0 ||
!same_copy_family(target.vkFormat, auroraFormat)) {
if (!valid(targets[eye])) {
return false;
}
}
for (uint32_t eye = 0; eye < targetCount; ++eye) {
m_targets[eye] = {
.buffer = targets[eye].buffer,
.width = targets[eye].width,
.height = targets[eye].height,
.vkFormat = targets[eye].vkFormat,
.acquireFenceFd = targets[eye].acquireFenceFd,
.acquireImageLayout = targets[eye].acquireImageLayout,
};
if (panel != nullptr && !valid(*panel)) {
return false;
}
for (uint32_t eye = targetCount; eye < m_targets.size(); ++eye) {
m_targets[eye] = {};
m_targets = {};
m_imageCount = 0;
const auto add = [&](uint32_t index, const AuroraVulkanStereoTarget& target) {
m_targets[index] = {
.buffer = target.buffer,
.width = target.width,
.height = target.height,
.vkFormat = target.vkFormat,
.acquireFenceFd = target.acquireFenceFd,
.acquireImageLayout = target.acquireImageLayout,
};
m_images[m_imageCount++] = index;
};
for (uint32_t eye = 0; eye < targetCount; ++eye) {
add(eye, targets[eye]);
}
if (panel != nullptr) {
add(kPanelIndex, *panel);
}
m_frameToken = token;
m_targetCount = targetCount;
@@ -202,7 +219,7 @@ public:
if (!m_framePending || !m_encoded || frame.frameToken != m_frameToken) {
return;
}
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS> releases{};
Releases releases{};
const bool success = EndAccessLocked(releases);
NotifyLocked(frame.frameToken, success, true, releases);
ClearFrameLocked();
@@ -215,7 +232,7 @@ public:
}
const uint64_t token = m_frameToken;
const bool encoded = m_encoded;
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS> releases{};
Releases releases{};
if (encoded) {
EndAccessLocked(releases);
for (auto& release : releases) {
@@ -242,7 +259,7 @@ private:
const auto& target = m_targets[eye];
if (source.texture == nullptr || source.format != m_auroraFormat ||
source.size.width != target.width || source.size.height != target.height) {
Log.error("Stereo eye {} does not match its OpenXR Vulkan target ({}x{} vs {}x{})", eye,
Log.error("Stereo image {} does not match its OpenXR Vulkan target ({}x{} vs {}x{})", eye,
source.size.width, source.size.height, target.width, target.height);
return nullptr;
}
@@ -264,7 +281,9 @@ private:
ahb.handle = target.buffer;
const wgpu::SharedTextureMemoryDescriptor memoryDescriptor{
.nextInChain = &ahb,
.label = eye == 0 ? "OpenXR left eye AHardwareBuffer" : "OpenXR right eye AHardwareBuffer",
.label = eye == 0 ? "OpenXR left eye AHardwareBuffer"
: eye == 1 ? "OpenXR right eye AHardwareBuffer"
: "OpenXR panel AHardwareBuffer",
};
import.memory = webgpu::g_device.ImportSharedTextureMemory(&memoryDescriptor);
if (!import.memory) {
@@ -291,7 +310,9 @@ private:
return nullptr;
}
const wgpu::TextureDescriptor textureDescriptor{
.label = eye == 0 ? "OpenXR left eye shared texture" : "OpenXR right eye shared texture",
.label = eye == 0 ? "OpenXR left eye shared texture"
: eye == 1 ? "OpenXR right eye shared texture"
: "OpenXR panel shared texture",
.usage = wgpu::TextureUsage::CopyDst,
.dimension = wgpu::TextureDimension::e2D,
.size = {target.width, target.height, 1},
@@ -313,19 +334,29 @@ private:
}
bool EncodeLocked(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) noexcept {
std::array<Import*, AURORA_VULKAN_STEREO_MAX_TARGETS> imports{};
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
imports[eye] = EnsureImport(eye, frame.eyes[eye]);
std::array<stereo::EyeImage, kMaxImages> sources{};
std::array<Import*, kMaxImages> imports{};
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
if (eye == kPanelIndex) {
if (!stereo_overlay::layer_source(encoder, m_targets[eye].width, m_targets[eye].height, sources[eye])) {
return false;
}
} else {
sources[eye] = frame.eyes[eye];
}
imports[eye] = EnsureImport(eye, sources[eye]);
if (imports[eye] == nullptr) {
return false;
}
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
auto& target = m_targets[eye];
auto& import = *imports[eye];
if (import.accessBegun) {
Log.error("AHardwareBuffer for eye {} is still under a previous access", eye);
RollbackAccesses(imports, eye);
RollbackAccesses(imports, n);
return false;
}
// The OpenXR side's release barrier leaves the image in acquireImageLayout;
@@ -350,7 +381,7 @@ private:
close_fd(target.acquireFenceFd);
if (!acquireFence) {
Log.error("Dawn could not import the OpenXR copy-out fence for eye {}", eye);
RollbackAccesses(imports, eye);
RollbackAccesses(imports, n);
return false;
}
}
@@ -370,15 +401,16 @@ private:
}
if (import.memory.BeginAccess(import.texture, &begin) != wgpu::Status::Success) {
Log.error("Dawn BeginAccess failed for stereo eye {}", eye);
RollbackAccesses(imports, eye);
RollbackAccesses(imports, n);
return false;
}
import.accessBegun = true;
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
const auto& import = *imports[eye];
const wgpu::TexelCopyTextureInfo source{
.texture = *frame.eyes[eye].texture,
.texture = *sources[eye].texture,
.mipLevel = 0,
.origin = {},
.aspect = wgpu::TextureAspect::All,
@@ -396,9 +428,10 @@ private:
return true;
}
void RollbackAccesses(const std::array<Import*, AURORA_VULKAN_STEREO_MAX_TARGETS>& imports,
uint32_t count) noexcept {
for (uint32_t eye = 0; eye < count; ++eye) {
// Ends the accesses begun for the first `count` images of this frame.
void RollbackAccesses(const std::array<Import*, kMaxImages>& imports, uint32_t count) noexcept {
for (uint32_t n = 0; n < count; ++n) {
const uint32_t eye = m_images[n];
if (imports[eye] != nullptr && imports[eye]->accessBegun) {
wgpu::SharedTextureMemoryEndAccessState end{};
imports[eye]->memory.EndAccess(imports[eye]->texture, &end);
@@ -411,13 +444,15 @@ private:
}
}
bool EndAccessLocked(
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS>& releases) noexcept {
// Fills one release per image of this frame, in order: the eyes, then the panel.
bool EndAccessLocked(Releases& releases) noexcept {
bool success = true;
for (auto& release : releases) {
release = {.releaseFenceFd = -1, .releasedImageLayout = VK_IMAGE_LAYOUT_UNDEFINED};
}
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
for (uint32_t n = 0; n < m_imageCount; ++n) {
const uint32_t eye = m_images[n];
auto& release = releases[n];
Import* import = m_encodedImports[eye];
if (import == nullptr || !import->accessBegun) {
success = false;
@@ -431,7 +466,7 @@ private:
success = false;
} else {
import->initialized = end.initialized;
releases[eye].releasedImageLayout = layout.newLayout;
release.releasedImageLayout = layout.newLayout;
for (size_t i = 0; i < end.fenceCount; ++i) {
wgpu::SharedFenceSyncFDExportInfo syncFd{};
wgpu::SharedFenceExportInfo info{};
@@ -441,16 +476,16 @@ private:
// The fence keeps its descriptor; hand the caller an independent one.
const int duplicate = ::dup(syncFd.handle);
if (duplicate >= 0) {
if (releases[eye].releaseFenceFd >= 0) {
if (release.releaseFenceFd >= 0) {
// Dawn normally returns exactly one fence per access. Both must
// be honoured and one descriptor cannot express two fences, so
// the earlier one is retired on the CPU before handing over the
// latest.
Log.warn("Dawn returned several release fences for eye {}; merging on the CPU", eye);
wait_sync_fd(releases[eye].releaseFenceFd);
close_fd(releases[eye].releaseFenceFd);
wait_sync_fd(release.releaseFenceFd);
close_fd(release.releaseFenceFd);
}
releases[eye].releaseFenceFd = duplicate;
release.releaseFenceFd = duplicate;
}
} else {
Log.error("Dawn returned a non-sync-fd fence for eye {}", eye);
@@ -482,12 +517,13 @@ private:
m_encodedImports = {};
m_frameToken = 0;
m_targetCount = 0;
m_imageCount = 0;
m_framePending = false;
m_encoded = false;
}
void PublishAndClearFrameLocked(uint64_t token, bool success, bool gpuWorkQueued) noexcept {
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS> releases{};
Releases releases{};
for (auto& release : releases) {
release = {.releaseFenceFd = -1, .releasedImageLayout = VK_IMAGE_LAYOUT_UNDEFINED};
}
@@ -496,10 +532,9 @@ private:
}
void NotifyLocked(uint64_t token, bool success, bool gpuWorkQueued,
const std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS>&
releases) noexcept {
const Releases& releases) noexcept {
if (m_callback != nullptr) {
m_callback(token, success, gpuWorkQueued, releases.data(), m_targetCount, m_userdata);
m_callback(token, success, gpuWorkQueued, releases.data(), m_imageCount, m_userdata);
} else {
for (auto release : releases) {
close_fd(release.releaseFenceFd);
@@ -509,8 +544,11 @@ private:
std::mutex m_mutex;
std::unordered_map<AHardwareBuffer*, Import> m_imports;
std::array<PendingTarget, AURORA_VULKAN_STEREO_MAX_TARGETS> m_targets{};
std::array<Import*, AURORA_VULKAN_STEREO_MAX_TARGETS> m_encodedImports{};
std::array<PendingTarget, kMaxImages> m_targets{};
std::array<Import*, kMaxImages> m_encodedImports{};
// The slots of m_targets this frame copies into, eyes first.
std::array<uint32_t, kMaxImages> m_images{};
uint32_t m_imageCount = 0;
wgpu::TextureFormat m_auroraFormat = wgpu::TextureFormat::Undefined;
AuroraVulkanStereoSubmittedCallback m_callback = nullptr;
void* m_userdata = nullptr;
@@ -575,7 +613,13 @@ bool aurora_vulkan_set_stereo_targets(uint64_t frameToken,
const AuroraVulkanStereoTarget* targets,
uint32_t targetCount) {
using namespace aurora::vulkan_interop;
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount);
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, nullptr);
}
bool aurora_vulkan_set_stereo_targets_with_panel(uint64_t frameToken, const AuroraVulkanStereoTarget* targets,
uint32_t targetCount, const AuroraVulkanStereoTarget* panel) {
using namespace aurora::vulkan_interop;
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, panel);
}
bool aurora_vulkan_cancel_stereo_targets(uint64_t frameToken) {
@@ -615,6 +659,11 @@ bool aurora_vulkan_set_stereo_targets(uint64_t, const AuroraVulkanStereoTarget*,
return false;
}
bool aurora_vulkan_set_stereo_targets_with_panel(uint64_t, const AuroraVulkanStereoTarget*, uint32_t,
const AuroraVulkanStereoTarget*) {
return false;
}
bool aurora_vulkan_cancel_stereo_targets(uint64_t) { return false; }
bool aurora_vulkan_disable_stereo_bridge() { return true; }
@@ -0,0 +1,220 @@
// SPDX-License-Identifier: GPL-3.0-or-later
#include <aurora/vulkan_win32_interop.h>
#include "../internal.hpp"
#include "../stereo.hpp"
#include "../stereo_overlay.hpp"
#include "gpu.hpp"
#if defined(_WIN32) && defined(WEBGPU_DAWN) && defined(DAWN_ENABLE_BACKEND_VULKAN)
#include <windows.h>
#include <algorithm>
#include <array>
#include <memory>
#include <mutex>
#include <vector>
namespace aurora::vulkan_win32 {
namespace {
Module Log("aurora::vulkan_win32");
struct Api {
AuroraDawnVulkanConfigureFn configure = nullptr;
AuroraDawnVulkanHandlesFn handles = nullptr;
AuroraDawnVulkanWrapFn wrap = nullptr;
AuroraDawnVulkanReleaseFn release = nullptr;
AuroraDawnVulkanLockFn lock = nullptr;
AuroraDawnVulkanUnlockFn unlock = nullptr;
AuroraDawnVulkanDrainFn drain = nullptr;
bool Load() {
HMODULE module = GetModuleHandleW(L"webgpu_dawn.dll");
if (!module) return false;
auto version = reinterpret_cast<AuroraDawnVulkanVersionFn>(GetProcAddress(module, "AuroraDawnVulkanVersion"));
if (!version || version() != AURORA_DAWN_VULKAN_ABI) return false;
#define LOAD(member, type, name) member = reinterpret_cast<type>(GetProcAddress(module, name)); if (!member) return false
LOAD(configure, AuroraDawnVulkanConfigureFn, "AuroraDawnVulkanConfigure");
LOAD(handles, AuroraDawnVulkanHandlesFn, "AuroraDawnVulkanGetHandles");
LOAD(wrap, AuroraDawnVulkanWrapFn, "AuroraDawnVulkanWrap");
LOAD(release, AuroraDawnVulkanReleaseFn, "AuroraDawnVulkanRelease");
LOAD(lock, AuroraDawnVulkanLockFn, "AuroraDawnVulkanLock");
LOAD(unlock, AuroraDawnVulkanUnlockFn, "AuroraDawnVulkanUnlock");
LOAD(drain, AuroraDawnVulkanDrainFn, "AuroraDawnVulkanDrain");
#undef LOAD
return true;
}
} api;
int64_t VkFormat(wgpu::TextureFormat format) {
switch (format) {
case wgpu::TextureFormat::RGBA8Unorm: return 37;
case wgpu::TextureFormat::RGBA8UnormSrgb: return 43;
case wgpu::TextureFormat::BGRA8Unorm: return 44;
case wgpu::TextureFormat::BGRA8UnormSrgb: return 50;
case wgpu::TextureFormat::RGBA16Float: return 97;
default: return 0;
}
}
bool CopyCompatible(int64_t a, int64_t b) {
return a == b || ((a == 37 || a == 43) && (b == 37 || b == 43)) ||
((a == 44 || a == 50) && (b == 44 || b == 50));
}
struct Import { uint64_t image; uint32_t width, height; wgpu::TextureFormat format; wgpu::Texture texture; };
// The eyes, then the settings panel's layer image after them.
constexpr uint32_t kMaxImages = 3;
class Bridge {
public:
std::mutex mutex;
std::vector<Import> imports;
std::array<AuroraD3D12StereoTarget, kMaxImages> targets{};
std::array<wgpu::Texture, kMaxImages> active{};
uint64_t token = 0;
uint32_t count = 0;
// Images this frame copies: `count` eyes, plus the panel when `panel` is set.
uint32_t images = 0;
bool panel = false;
bool encoded = false;
AuroraD3D12StereoSubmittedCallback callback;
void* userdata;
Bridge(AuroraD3D12StereoSubmittedCallback cb, void* data) : callback(cb), userdata(data) {}
bool Set(uint64_t next, const AuroraD3D12StereoTarget* data, uint32_t n, const AuroraD3D12StereoTarget* panelTarget) {
if (!next || !data || !n || n > 2) return false;
std::lock_guard guard(mutex);
if (token) return false;
const auto valid = [](const AuroraD3D12StereoTarget& target) {
return target.resource && target.width && target.height;
};
for (uint32_t i = 0; i < n; ++i) {
if (!valid(data[i])) return false;
}
if (panelTarget && !valid(*panelTarget)) return false;
for (uint32_t i = 0; i < n; ++i) targets[i] = data[i];
panel = panelTarget != nullptr;
if (panel) targets[n] = *panelTarget;
token = next; count = n; images = n + (panel ? 1u : 0u); return true;
}
bool Cancel(uint64_t wanted) {
std::unique_lock guard(mutex, std::try_to_lock);
if (!guard.owns_lock() || encoded || !token || token != wanted) return false;
token = 0; return true;
}
bool Encode(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) {
std::lock_guard guard(mutex);
if (!token || encoded || frame.frameToken != token) return false;
std::array<stereo::EyeImage, kMaxImages> sources{};
for (uint32_t i = 0; i < count; ++i) sources[i] = frame.eyes[i];
if (panel && !stereo_overlay::layer_source(encoder, targets[count].width, targets[count].height, sources[count]))
return false;
// Validate/import every target before recording any copy.
for (uint32_t i = 0; i < images; ++i) {
const auto& eye = sources[i];
const auto& target = targets[i];
if (!eye.texture || eye.size.width != target.width || eye.size.height != target.height ||
!CopyCompatible(VkFormat(eye.format), target.dxgiFormat)) return false;
const uint64_t image = reinterpret_cast<uintptr_t>(target.resource);
auto it = std::find_if(imports.begin(), imports.end(), [&](const Import& entry) { return entry.image == image; });
if (it == imports.end()) {
wgpu::TextureDescriptor descriptor;
descriptor.usage = wgpu::TextureUsage::CopyDst | wgpu::TextureUsage::RenderAttachment;
descriptor.size = {target.width, target.height, 1};
descriptor.format = eye.format;
void* wrapped = api.wrap(webgpu::g_device.Get(), &descriptor, image);
if (!wrapped) return false;
imports.push_back({image, target.width, target.height, eye.format,
wgpu::Texture::Acquire(static_cast<WGPUTexture>(wrapped))});
it = imports.end() - 1;
}
if (it->width != target.width || it->height != target.height || it->format != eye.format) {
// The runtime reuses VkImage handles across swapchain recreation, and
// Aurora's eye format can change with the surface. Re-wrap rather than
// rejecting every future frame for this image.
imports.erase(it);
--i;
continue;
}
active[i] = it->texture;
}
for (uint32_t i = 0; i < images; ++i) {
wgpu::TexelCopyTextureInfo source, destination;
source.texture = *sources[i].texture;
destination.texture = active[i];
wgpu::Extent3D size{targets[i].width, targets[i].height, 1};
encoder.CopyTextureToTexture(&source, &destination, &size);
}
encoded = true;
return true;
}
void Submitted(const stereo::SinkFrame& frame) {
std::lock_guard guard(mutex);
if (!token || !encoded || token != frame.frameToken) return;
std::array<void*, kMaxImages> textures{};
for (uint32_t i = 0; i < images; ++i) textures[i] = active[i].Get();
// Append the COLOR_ATTACHMENT_OPTIMAL release barriers to Dawn's queue,
// flush them under its device guard, then allow the XR thread to release.
const bool success = api.release(webgpu::g_device.Get(), textures.data(), images) != 0;
const auto completed = token;
token = 0; encoded = false;
callback(completed, success, userdata);
}
};
std::unique_ptr<Bridge> bridge;
}
}
bool aurora_vulkan_win32_configure(const AuroraDawnVulkanHooks* hooks) {
using namespace aurora::vulkan_win32;
return api.Load() && api.configure(hooks);
}
bool aurora_vulkan_win32_get_handles(AuroraDawnVulkanHandles* handles, int64_t* format) {
using namespace aurora;
if (!webgpu::g_device || webgpu::g_backendType != wgpu::BackendType::Vulkan || !vulkan_win32::api.Load()) return false;
*format = vulkan_win32::VkFormat(webgpu::g_graphicsConfig.surfaceConfiguration.format);
return *format && vulkan_win32::api.handles(webgpu::g_device.Get(), handles);
}
bool aurora_vulkan_win32_enable(AuroraD3D12StereoSubmittedCallback cb, void* data) {
using namespace aurora::vulkan_win32;
if (bridge || !cb || !api.Load()) return false;
bridge = std::make_unique<Bridge>(cb, data);
aurora::stereo::set_sink(
[](wgpu::CommandEncoder& encoder, const aurora::stereo::SinkFrame& frame, void* self) noexcept {
return static_cast<Bridge*>(self)->Encode(encoder, frame);
}, [](const aurora::stereo::SinkFrame& frame, void* self) noexcept { static_cast<Bridge*>(self)->Submitted(frame); }, bridge.get());
return true;
}
bool aurora_vulkan_win32_set_targets(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count) {
using namespace aurora::vulkan_win32;
return bridge && bridge->Set(token, targets, count, nullptr);
}
bool aurora_vulkan_win32_set_targets_with_panel(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count,
const AuroraD3D12StereoTarget* panel) {
using namespace aurora::vulkan_win32;
return bridge && bridge->Set(token, targets, count, panel);
}
bool aurora_vulkan_win32_cancel(uint64_t token) {
using namespace aurora::vulkan_win32;
return bridge && bridge->Cancel(token);
}
bool aurora_vulkan_win32_disable() {
using namespace aurora::vulkan_win32;
if (!bridge) return true;
aurora::stereo::set_sink(nullptr, nullptr, nullptr);
if (!api.drain(aurora::webgpu::g_device.Get())) { bridge.release(); return false; }
bridge.reset();
return true;
}
void* aurora_vulkan_win32_lock_queue() {
return aurora::vulkan_win32::api.lock(aurora::webgpu::g_device.Get());
}
void aurora_vulkan_win32_unlock_queue(void* guard) { aurora::vulkan_win32::api.unlock(guard); }
#else
// C ABI stubs keep the runtime's OpenXR integration linkable on Windows GX
// builds whose Dawn has no Vulkan backend; the backend then reports that the
// bridge is unavailable and the game falls back to the desktop renderer.
bool aurora_vulkan_win32_configure(const AuroraDawnVulkanHooks*) { return false; }
bool aurora_vulkan_win32_get_handles(AuroraDawnVulkanHandles* handles, int64_t* format) {
if (handles) *handles = {};
if (format) *format = 0;
return false;
}
bool aurora_vulkan_win32_enable(AuroraD3D12StereoSubmittedCallback, void*) { return false; }
bool aurora_vulkan_win32_set_targets(uint64_t, const AuroraD3D12StereoTarget*, uint32_t) { return false; }
bool aurora_vulkan_win32_set_targets_with_panel(uint64_t, const AuroraD3D12StereoTarget*, uint32_t,
const AuroraD3D12StereoTarget*) { return false; }
bool aurora_vulkan_win32_cancel(uint64_t) { return false; }
bool aurora_vulkan_win32_disable() { return true; }
void* aurora_vulkan_win32_lock_queue() { return nullptr; }
void aurora_vulkan_win32_unlock_queue(void*) {}
#endif
+38
View File
@@ -0,0 +1,38 @@
"""Apply the versioned Aurora native Vulkan ABI to the pinned Dawn source tree."""
from pathlib import Path
import shutil, sys
root = Path(sys.argv[1]).resolve()
here = Path(__file__).resolve().parent
for name in ("aurora_vulkan_hooks.h", "aurora_vulkan_interop.inc"):
shutil.copyfile(here / name, root / "src/dawn/native/vulkan" / name)
shutil.copyfile(here.parent.parent / "include/aurora/dawn_vulkan_abi.h",
root / "src/dawn/native/vulkan/aurora_dawn_vulkan_abi.h")
p = root / "src/dawn/native/vulkan/VulkanBackend.cpp"
s = p.read_text()
if '#include "aurora_vulkan_interop.inc"' not in s:
s += '\n#include "aurora_vulkan_interop.inc"\n'
p.write_text(s)
p = root / "src/dawn/common/DynamicLib.cpp"
s = p.read_text()
# LOAD_LIBRARY_SEARCH_DLL_LOAD_DIR requires an absolute filename. Vulkan's
# system loader is also tried by bare name; preserve the restricted default
# search directories for that case instead of failing with ERROR_INVALID_PARAMETER.
old = 'LOAD_LIBRARY_SEARCH_DLL_LOAD_DIR | LOAD_LIBRARY_SEARCH_DEFAULT_DIRS;'
new = '''((filename.size() > 2 && filename[1] == ':') || filename.starts_with("\\\\\\\\")
? LOAD_LIBRARY_SEARCH_DLL_LOAD_DIR : 0) | LOAD_LIBRARY_SEARCH_DEFAULT_DIRS;'''
if old in s:
s = s.replace(old, new, 1)
p.write_text(s)
p = root / "src/dawn/native/vulkan/VulkanFunctions.cpp"
s = p.read_text()
if '#include "aurora_vulkan_hooks.h"' not in s:
marker = 'namespace dawn::native::vulkan {'
assert marker in s
s = s.replace(marker, '#include "aurora_vulkan_hooks.h"\n\n' + marker, 1)
for old, new in [
('GET_GLOBAL_PROC(CreateInstance);', 'CreateInstance = AuroraCreateInstance(GetInstanceProcAddr);'),
('GET_INSTANCE_PROC(CreateDevice);', 'CreateDevice = AuroraCreateDevice(GetInstanceProcAddr, instance);'),
('GET_INSTANCE_PROC(EnumeratePhysicalDevices);', 'EnumeratePhysicalDevices = AuroraEnumeratePhysicalDevices(GetInstanceProcAddr, instance);')]:
assert old in s, old
s = s.replace(old, old + '\n ' + new, 1)
p.write_text(s)
@@ -0,0 +1,9 @@
// Included only by the pinned Dawn source build.
#pragma once
#include "dawn/native/DawnNative.h"
#include "aurora_dawn_vulkan_abi.h"
namespace dawn::native::vulkan {
PFN_vkCreateInstance AuroraCreateInstance(PFN_vkGetInstanceProcAddr proc);
PFN_vkCreateDevice AuroraCreateDevice(PFN_vkGetInstanceProcAddr proc, VkInstance instance);
PFN_vkEnumeratePhysicalDevices AuroraEnumeratePhysicalDevices(PFN_vkGetInstanceProcAddr proc, VkInstance instance);
}
@@ -0,0 +1,118 @@
// Compiled inside VulkanBackend.cpp, using the pinned Dawn implementation.
#include "aurora_vulkan_hooks.h"
#include "src/dawn/native/ChainUtils.h"
#include "src/dawn/native/vulkan/PhysicalDeviceVk.h"
#include "src/dawn/native/vulkan/QueueVk.h"
#include <mutex>
namespace dawn::native::vulkan {
namespace {
AuroraDawnVulkanHooks auroraHooks{};
std::recursive_mutex auroraHooksMutex;
PFN_vkGetInstanceProcAddr auroraGetProc = nullptr;
VkInstance auroraInstance = VK_NULL_HANDLE;
::VkResult VKAPI_CALL AuroraCreateInstanceImpl(const VkInstanceCreateInfo* info,
const VkAllocationCallbacks* allocator, VkInstance* instance) {
std::lock_guard lock(auroraHooksMutex);
if (auroraHooks.createInstance) return static_cast<::VkResult>(auroraHooks.createInstance(
auroraHooks.userdata, reinterpret_cast<void*>(auroraGetProc), info, allocator,
reinterpret_cast<void**>(instance)));
return reinterpret_cast<PFN_vkCreateInstance>(auroraGetProc(nullptr, "vkCreateInstance"))(info, allocator, instance);
}
::VkResult VKAPI_CALL AuroraCreateDeviceImpl(VkPhysicalDevice physical, const VkDeviceCreateInfo* info,
const VkAllocationCallbacks* allocator, VkDevice* device) {
std::lock_guard lock(auroraHooksMutex);
if (auroraHooks.createDevice) return static_cast<::VkResult>(auroraHooks.createDevice(
auroraHooks.userdata, reinterpret_cast<void*>(auroraGetProc), physical, info, allocator,
reinterpret_cast<void**>(device)));
return reinterpret_cast<PFN_vkCreateDevice>(auroraGetProc(auroraInstance, "vkCreateDevice"))(physical, info, allocator, device);
}
::VkResult VKAPI_CALL AuroraEnumerateImpl(VkInstance instance, uint32_t* count, VkPhysicalDevice* devices) {
std::lock_guard lock(auroraHooksMutex);
if (!auroraHooks.getPhysicalDevice) return reinterpret_cast<PFN_vkEnumeratePhysicalDevices>(
auroraGetProc(instance, "vkEnumeratePhysicalDevices"))(instance, count, devices);
void* physical = nullptr;
const auto result = static_cast<::VkResult>(auroraHooks.getPhysicalDevice(auroraHooks.userdata, instance, &physical));
if (result != VK_SUCCESS) return result;
if (!devices) { *count = 1; return VK_SUCCESS; }
if (*count == 0) return VK_INCOMPLETE;
devices[0] = static_cast<VkPhysicalDevice>(physical);
*count = 1;
return VK_SUCCESS;
}
struct AuroraDeviceGuard { decltype(std::declval<Device*>()->GetGuard()) guard; explicit AuroraDeviceGuard(Device* device) : guard(device->GetGuard()) {} };
}
PFN_vkCreateInstance AuroraCreateInstance(PFN_vkGetInstanceProcAddr proc) {
std::lock_guard lock(auroraHooksMutex);
if (!auroraHooks.createInstance) return reinterpret_cast<PFN_vkCreateInstance>(proc(nullptr, "vkCreateInstance"));
auroraGetProc = proc;
return AuroraCreateInstanceImpl;
}
PFN_vkCreateDevice AuroraCreateDevice(PFN_vkGetInstanceProcAddr proc, VkInstance instance) {
std::lock_guard lock(auroraHooksMutex);
if (!auroraHooks.createDevice) return reinterpret_cast<PFN_vkCreateDevice>(proc(instance, "vkCreateDevice"));
auroraGetProc = proc; auroraInstance = instance;
return AuroraCreateDeviceImpl;
}
PFN_vkEnumeratePhysicalDevices AuroraEnumeratePhysicalDevices(PFN_vkGetInstanceProcAddr proc, VkInstance instance) {
std::lock_guard lock(auroraHooksMutex);
if (!auroraHooks.getPhysicalDevice) return reinterpret_cast<PFN_vkEnumeratePhysicalDevices>(proc(instance, "vkEnumeratePhysicalDevices"));
auroraGetProc = proc; auroraInstance = instance;
return AuroraEnumerateImpl;
}
extern "C" DAWN_NATIVE_EXPORT uint32_t AuroraDawnVulkanVersion() { return AURORA_DAWN_VULKAN_ABI; }
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanConfigure(const AuroraDawnVulkanHooks* hooks) {
std::lock_guard lock(auroraHooksMutex);
// Called before adapter discovery, or cleared after Aurora shutdown. The
// runtime and hook storage must outlive all device creation calls.
auroraHooks = hooks ? *hooks : AuroraDawnVulkanHooks{};
return 1;
}
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanGetHandles(void* handle, AuroraDawnVulkanHandles* out) {
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
auto guard = device->GetGuard();
if (!device->HasFeature(Feature::ImplicitDeviceSynchronization)) return 0;
*out = {device->GetVkInstance(), ToBackend(device->GetPhysicalDevice())->GetVkPhysicalDevice(),
device->GetVkDevice(), device->GetGraphicsQueueFamily(), 0};
return 1;
}
extern "C" DAWN_NATIVE_EXPORT void* AuroraDawnVulkanWrap(void* handle, const void* desc, uint64_t image) {
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
auto guard = device->GetGuard();
// Mirror DeviceBase::CreateTexture: fill the trivial frontend defaults
// (dimension, mip and sample counts) and validate before the backend reads
// them. An Undefined dimension would otherwise abort the first copy.
TextureDescriptor raw = reinterpret_cast<const TextureDescriptor*>(desc)->WithTrivialFrontendDefaults();
UnpackedPtr<TextureDescriptor> descriptor;
if (device->ConsumedError(ValidateAndUnpack(&raw), &descriptor)) return nullptr;
if (device->ConsumedError(ValidateTextureDescriptor(device, descriptor, AllowMultiPlanarTextureFormat::No))) return nullptr;
auto texture = SwapChainTexture::Create(device, descriptor, VkImage::CreateFromHandle(reinterpret_cast<::VkImage>(image)));
texture->SetIsSubresourceContentInitialized(true, texture->GetAllSubresources());
texture->UpdateUsage(wgpu::TextureUsage::RenderAttachment, wgpu::ShaderStage::None, texture->GetAllSubresources());
return ToAPI(ReturnToAPI(std::move(texture)));
}
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanRelease(void* handle, void* const* textures, uint32_t count) {
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
auto guard = device->GetGuard();
auto* queue = ToBackend(device->GetQueue());
auto* context = queue->GetPendingRecordingContext();
for (uint32_t i = 0; i < count; ++i) {
auto* texture = ToBackend(FromAPI(static_cast<WGPUTexture>(textures[i])));
texture->TransitionUsageNow(context, wgpu::TextureUsage::RenderAttachment,
wgpu::ShaderStage::None, texture->GetAllSubresources());
}
return !device->ConsumedError(queue->SubmitPendingCommands());
}
extern "C" DAWN_NATIVE_EXPORT void* AuroraDawnVulkanLock(void* handle) {
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
return new AuroraDeviceGuard(device);
}
extern "C" DAWN_NATIVE_EXPORT void AuroraDawnVulkanUnlock(void* guard) {
delete static_cast<AuroraDeviceGuard*>(guard);
}
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanDrain(void* handle) {
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
auto guard = device->GetGuard();
if (device->ConsumedError(ToBackend(device->GetQueue())->SubmitPendingCommands())) return 0;
return device->fn.QueueWaitIdle(ToBackend(device->GetQueue())->GetVkQueue()) == VK_SUCCESS;
}
}
+21
View File
@@ -3,6 +3,18 @@ include(GoogleTest)
option(AURORA_GPU_SMOKE_TESTS "Build opt-in tests requiring a desktop GPU" OFF)
if (AURORA_GPU_SMOKE_TESTS AND AURORA_ENABLE_GX AND WIN32)
# Exercises the custom Dawn DLL's Aurora Vulkan ABI (patches/dawn). Run it with that
# webgpu_dawn.dll beside the executable; the stock prebuilt Dawn lacks the exports and the
# test reports that at startup. Vulkan headers come from the SDK or any tree on
# AURORA_TEST_VULKAN_INCLUDE (the runtime's FetchContent copy works).
find_path(AURORA_TEST_VULKAN_INCLUDE vulkan/vulkan.h HINTS "$ENV{VULKAN_SDK}/Include")
if (AURORA_TEST_VULKAN_INCLUDE)
add_executable(vulkan_native_bridge_smoke vulkan_native_bridge_smoke.cpp)
target_include_directories(vulkan_native_bridge_smoke PRIVATE ../include "${AURORA_TEST_VULKAN_INCLUDE}")
target_link_libraries(vulkan_native_bridge_smoke PRIVATE dawn::webgpu_dawn dawn::dawncpp_headers)
else ()
message(STATUS "vulkan_native_bridge_smoke skipped: no vulkan/vulkan.h (set VULKAN_SDK or AURORA_TEST_VULKAN_INCLUDE)")
endif ()
add_executable(stereo_frame_worker_smoke stereo_frame_worker_smoke.cpp)
target_include_directories(stereo_frame_worker_smoke PRIVATE ../lib)
target_link_libraries(stereo_frame_worker_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi
@@ -15,6 +27,13 @@ if (AURORA_GPU_SMOKE_TESTS AND AURORA_ENABLE_GX AND WIN32)
target_include_directories(efb_ram_lifetime_smoke PRIVATE ../lib)
target_link_libraries(efb_ram_lifetime_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi
dawn::dawncpp_headers)
# VR cockpit overlay (hands, synthetic wheel) against real scene depth. Standalone: it
# defines the GPU globals itself and needs only the header.
add_executable(cockpit_gpu_smoke cockpit_gpu_smoke.cpp)
target_include_directories(cockpit_gpu_smoke PRIVATE ../include ../lib)
target_compile_definitions(cockpit_gpu_smoke PRIVATE AURORA TARGET_PC WEBGPU_DAWN)
target_link_libraries(cockpit_gpu_smoke PRIVATE fmt::fmt xxhash absl::flat_hash_map absl::btree
dawn::webgpu_dawn dawn::dawncpp_headers TracyClient ${AURORA_SDL3_TARGET})
endif ()
if (NOT TARGET gtest)
@@ -36,6 +55,8 @@ if (AURORA_ENABLE_GX)
stereo_replay_test.cpp
stereo_interpolation_test.cpp
stereo_mirror_test.cpp
native_wheel_test.cpp
cockpit_geometry_test.cpp
texture_bind_group_cache_key_test.cpp
../lib/gfx/efb_ram_encoder.cpp
# GX API implementations (encoders)
+260
View File
@@ -0,0 +1,260 @@
// SPDX-License-Identifier: GPL-3.0-or-later
// VR cockpit overlay geometry: what the synthetic wheel, handlebar and hands
// build in the seated frame, without a GPU.
#include <gtest/gtest.h>
#include <cstring>
#include "gfx/cockpit.hpp"
namespace {
using aurora::gfx::cockpit::V;
using aurora::gfx::cockpit::Vertex;
bool all_finite(const std::vector<Vertex>& vertices) {
for (const auto& vertex : vertices) {
for (float value : vertex.position) {
uint32_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
if ((bits & 0x7f800000u) == 0x7f800000u) {
return false;
}
}
}
return true;
}
void set_identity(float (&matrix)[12], V translation) {
const auto identity = aurora::gfx::cockpit::identity();
std::memcpy(matrix, identity.data(), sizeof(matrix));
matrix[3] = translation[0];
matrix[7] = translation[1];
matrix[11] = translation[2];
}
class CockpitGeometry : public ::testing::Test {
protected:
void SetUp() override { clear_meshes(); }
void TearDown() override { clear_meshes(); }
static void clear_meshes() {
std::lock_guard lock(aurora::gfx::cockpit::meshMutex);
aurora::gfx::cockpit::meshes = {};
}
};
TEST_F(CockpitGeometry, NativeWheelWithoutHandsDrawsNothing) {
AuroraCockpit cockpit{};
cockpit.nativeWheel = true;
EXPECT_TRUE(aurora::gfx::cockpit::geometry(cockpit).empty());
}
TEST_F(CockpitGeometry, SyntheticKartWheelSitsOnItsRim) {
AuroraCockpit cockpit{};
const auto vertices = aurora::gfx::cockpit::geometry(cockpit);
ASSERT_FALSE(vertices.empty());
ASSERT_TRUE(all_finite(vertices));
// The rim, spokes and hub stay within the 0.18 m wheel plus its tube, around
// the wheel centre the input side uses (steering_wheel.h).
for (const auto& vertex : vertices) {
const float x = vertex.position[0];
const float y = vertex.position[1] + 0.30f;
EXPECT_LE(std::hypot(x, y), 0.18f + 0.02f);
EXPECT_NEAR(vertex.position[2], -0.42f, 0.04f);
}
}
TEST_F(CockpitGeometry, SyntheticWheelTurnsWithTheAngle) {
AuroraCockpit cockpit{};
const auto straight = aurora::gfx::cockpit::geometry(cockpit);
cockpit.wheelAngle = 0.5f;
const auto turned = aurora::gfx::cockpit::geometry(cockpit);
ASSERT_EQ(straight.size(), turned.size());
bool moved = false;
for (size_t i = 0; i < straight.size() && !moved; ++i) {
moved = std::abs(straight[i].position[0] - turned[i].position[0]) > 1e-3f;
}
EXPECT_TRUE(moved);
}
TEST_F(CockpitGeometry, SyntheticHandlebarFollowsItsFrame) {
AuroraCockpit cockpit{};
cockpit.bike = true;
cockpit.handlebarRadius = 0.25f;
// Bar axis along seat +X, centred 0.3 m down and 0.42 m ahead.
const float pose[12]{1, 0, 0, 0, 0, 0, 1, -0.3f, 0, -1, 0, -0.42f};
std::memcpy(cockpit.seatFromHandlebar, pose, sizeof(pose));
const auto vertices = aurora::gfx::cockpit::geometry(cockpit);
ASSERT_FALSE(vertices.empty());
ASSERT_TRUE(all_finite(vertices));
float minX = 1e9f;
float maxX = -1e9f;
for (const auto& vertex : vertices) {
minX = std::min(minX, vertex.position[0]);
maxX = std::max(maxX, vertex.position[0]);
}
EXPECT_NEAR(minX, -0.25f, 0.03f);
EXPECT_NEAR(maxX, 0.25f, 0.03f);
}
TEST_F(CockpitGeometry, TrackedHandDrawsAGloveAtItsGrip) {
AuroraCockpit cockpit{};
cockpit.nativeWheel = true;
cockpit.hands[1].tracked = true;
cockpit.hands[1].squeeze = 1.0f;
set_identity(cockpit.hands[1].seatFromGrip, {0.2f, -0.3f, -0.4f});
const auto vertices = aurora::gfx::cockpit::geometry(cockpit);
ASSERT_FALSE(vertices.empty());
ASSERT_TRUE(all_finite(vertices));
for (const auto& vertex : vertices) {
EXPECT_LT(std::abs(vertex.position[0] - 0.2f), 0.15f);
EXPECT_LT(std::abs(vertex.position[1] + 0.3f), 0.15f);
EXPECT_LT(std::abs(vertex.position[2] + 0.4f), 0.15f);
}
}
// The grip space OpenXR defines: -Z up the curled fingers' tube towards the
// thumb, +X out of the palm. So the fingers run along Y (+Y on the right hand,
// -Y on the left) and close towards +X, never out of the back of the hand.
TEST_F(CockpitGeometry, GloveFingersRunAlongTheHandAndCloseIntoThePalm) {
// Both grips carry the same orientation when the hands hold a wheel symmetrically, so the fingers
// run along -Y on both and it is the palm side that mirrors: +X on the left hand, -X on the right.
// Building the right hand's fingers on +Y instead pointed them at the player (PC, 2026-09-23).
for (int side = 0; side < 2; ++side) {
const float palmSide = side == 0 ? 1.0f : -1.0f;
const auto build = [&](float squeeze) {
AuroraCockpit cockpit{};
cockpit.nativeWheel = true;
cockpit.hands[side].tracked = true;
cockpit.hands[side].squeeze = squeeze;
set_identity(cockpit.hands[side].seatFromGrip, {0.0f, 0.0f, 0.0f});
return aurora::gfx::cockpit::geometry(cockpit);
};
struct Extent {
float reach = 0.0f; // furthest along the fingers
float palm = 0.0f; // furthest towards the palm's normal
float back = 0.0f; // furthest out of the back of the hand
float across = 0.0f; // furthest across the knuckles
};
const auto measure = [&](const std::vector<Vertex>& vertices) {
Extent e{};
for (const auto& vertex : vertices) {
e.reach = std::max(e.reach, -vertex.position[1]);
e.palm = std::max(e.palm, vertex.position[0] * palmSide);
e.back = std::min(e.back, vertex.position[0] * palmSide);
e.across = std::max(e.across, std::abs(vertex.position[2]));
}
return e;
};
const auto open = measure(build(0.0f));
const auto closed = measure(build(1.0f));
EXPECT_GT(open.reach, 0.09f) << "open fingers reach along the hand, side " << side;
EXPECT_LT(open.palm, 0.05f) << "an open hand is flat, side " << side;
EXPECT_LT(closed.reach, open.reach - 0.02f) << "closing shortens the reach, side " << side;
EXPECT_GT(closed.palm, open.palm + 0.02f) << "closing moves the fingers into the palm, side " << side;
EXPECT_GT(closed.back, -0.03f) << "fingers never bend out of the back of the hand, side " << side;
EXPECT_LT(closed.across, 0.07f) << "fingers stay across the knuckles, side " << side;
}
}
TEST_F(CockpitGeometry, RuntimeFingersCurlTowardPalmForSqueezeAndWheelGrab) {
using namespace aurora::gfx::cockpit;
// OpenXR joint space: -Z runs toward the fingertip, +Y out of the back
// of the hand, for BOTH hands. Mirror positions, not the curl direction.
for (int side = 0; side < 2; ++side) {
SCOPED_TRACE(side);
HandMesh mesh;
mesh.parents.fill(1);
mesh.parents[1] = -1;
const float rootPose[7]{0, 0, 0.70710678f, 0.70710678f, 0.12f, -0.08f, 0.03f};
const M root = from_pose(rootPose);
mesh.bind.fill(root);
const int bases[]{2, 6, 11, 16, 21};
for (int finger = 0; finger < 5; ++finger) {
const int base = bases[finger];
const int count = finger == 0 ? 4 : 5;
for (int bone = 0; bone < count; ++bone) {
M bind = identity();
bind[3] = (side == 0 ? -1.0f : 1.0f) * (finger - 2) * 0.018f;
bind[11] = -0.025f * (bone + 1);
mesh.bind[base + bone] = compose(root, bind);
mesh.parents[base + bone] = bone == 0 ? 1 : base + bone - 1;
}
}
for (int j = 0; j < 26; ++j) mesh.inverseBind[j] = inverse(mesh.bind[j]);
// A tiny triangle rigidly weighted to each joint, including each fingertip.
for (int j = 0; j < 26; ++j) {
for (V offset : {V{0, 0, 0}, V{0.001f, 0, 0}, V{0, 0, 0.001f}}) {
AuroraVRHandVertex vertex{};
const V p = point(mesh.bind[j].data(), offset);
std::memcpy(vertex.position, p.data(), sizeof(vertex.position));
vertex.joints[0] = j;
vertex.weights[0] = 1;
mesh.indices.push_back(static_cast<uint16_t>(mesh.vertices.size()));
mesh.vertices.push_back(vertex);
}
}
AuroraCockpitHand hand{};
set_identity(hand.seatFromGrip, {0, 0, 0});
const auto build = [&](float squeeze, bool held) {
hand.squeeze = squeeze;
hand.held = held;
std::vector<Vertex> vertices;
runtime_hand(vertices, hand, mesh);
return vertices;
};
const auto open = build(0, false);
for (int j = 0; j < 26; ++j) {
const V bind = point(mesh.inverseBind[1].data(), point(mesh.bind[j].data(), {0, 0, 0}));
for (int axis = 0; axis < 3; ++axis)
EXPECT_NEAR(open[j * 3].position[axis], bind[axis] + (axis == 2 ? 0.04f : 0), 1e-6f);
}
for (const auto& closed : {build(0.5f, false), build(1, false), build(0, true)}) {
ASSERT_TRUE(all_finite(closed));
for (int tip : {5, 10, 15, 20, 25}) {
EXPECT_LT(closed[tip * 3].position[1], open[tip * 3].position[1] - 0.005f)
<< "fingertip must move toward palm (-Y), joint " << tip;
EXPECT_GT(closed[tip * 3].position[2], open[tip * 3].position[2])
<< "curl must shorten finger reach, joint " << tip;
}
for (int rigid : {0, 1, 6, 11, 16, 21})
for (int axis = 0; axis < 3; ++axis)
EXPECT_NEAR(closed[rigid * 3].position[axis], open[rigid * 3].position[axis], 1e-6f);
}
}
}
TEST_F(CockpitGeometry, RuntimeHandMeshIsSkinnedWithoutNans) {
using namespace aurora::gfx::cockpit;
auto mesh = std::make_shared<HandMesh>();
// A 26-joint chain, each joint 1 cm past its parent; one triangle on the tip.
for (int j = 0; j < 26; ++j) {
mesh->bind[j] = identity();
mesh->bind[j][11] = -0.01f * float(j);
mesh->inverseBind[j] = inverse(mesh->bind[j]);
mesh->parents[j] = j - 1;
}
for (int i = 0; i < 3; ++i) {
AuroraVRHandVertex vertex{};
vertex.position[0] = 0.01f * float(i);
vertex.position[2] = -0.25f;
vertex.joints[0] = 25;
vertex.joints[1] = vertex.joints[2] = vertex.joints[3] = -1;
vertex.weights[0] = 1.0f;
mesh->vertices.push_back(vertex);
mesh->indices.push_back(uint16_t(i));
}
{
std::lock_guard lock(meshMutex);
meshes[0] = mesh;
}
AuroraCockpit cockpit{};
cockpit.nativeWheel = true;
cockpit.hands[0].tracked = true;
cockpit.hands[0].held = true;
set_identity(cockpit.hands[0].seatFromGrip, {-0.2f, -0.3f, -0.4f});
const auto vertices = geometry(cockpit);
ASSERT_EQ(vertices.size(), 3u) << "the runtime mesh replaces the glove";
EXPECT_TRUE(all_finite(vertices));
}
} // namespace
+168
View File
@@ -0,0 +1,168 @@
// SPDX-License-Identifier: GPL-3.0-or-later
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
// Renders the VR cockpit overlay on a real GPU against cleared, occluding and
// partially occluding scene depth, forward and reversed, 1x and 4x MSAA.
#include "../lib/gfx/cockpit.hpp"
#include <fstream>
#include <iostream>
#include <atomic>
namespace aurora::webgpu { wgpu::Device g_device; wgpu::Queue g_queue; GraphicsConfig g_graphicsConfig{}; }
std::atomic<int> errors=0;
int main() {
using namespace aurora;
using namespace webgpu;
wgpu::InstanceDescriptor id{};
const wgpu::InstanceFeatureName timed=wgpu::InstanceFeatureName::TimedWaitAny;
id.requiredFeatureCount=1;id.requiredFeatures=&timed;
auto instance=wgpu::CreateInstance(&id);
wgpu::Adapter adapter;
wgpu::RequestAdapterOptions options{.backendType=wgpu::BackendType::D3D12};
auto future=instance.RequestAdapter(&options,wgpu::CallbackMode::WaitAnyOnly,
[&](wgpu::RequestAdapterStatus status,wgpu::Adapter a,wgpu::StringView message) {
if(status==wgpu::RequestAdapterStatus::Success) adapter=std::move(a);
else std::cerr<<std::string_view(message)<<'\n';
});
if(instance.WaitAny(future,5000000000)!=wgpu::WaitStatus::Success||!adapter) return 1;
wgpu::DeviceDescriptor dd{};
dd.SetUncapturedErrorCallback([](const wgpu::Device&,wgpu::ErrorType,wgpu::StringView message) {
++errors;std::cerr<<std::string_view(message)<<'\n';
});
future=adapter.RequestDevice(&dd,wgpu::CallbackMode::WaitAnyOnly,
[&](wgpu::RequestDeviceStatus status,wgpu::Device device,wgpu::StringView message) {
if(status==wgpu::RequestDeviceStatus::Success) g_device=std::move(device);
else std::cerr<<std::string_view(message)<<'\n';
});
if(instance.WaitAny(future,5000000000)!=wgpu::WaitStatus::Success||!g_device) return 1;
g_queue=g_device.GetQueue();
g_graphicsConfig.surfaceConfiguration.format=wgpu::TextureFormat::RGBA8Unorm;
g_graphicsConfig.depthFormat=wgpu::TextureFormat::Depth32Float;
AuroraCockpit native{}; native.nativeWheel=true;
if(!gfx::cockpit::geometry(native).empty()) return 1;
native.nativeWheel=false;
if(gfx::cockpit::geometry(native).empty()) return 1;
for(bool bike : {false,true}) for(bool original : {false,true}) for(uint32_t samples : {1u,4u})
for(bool hud : {false,true}) for(int coverage : {0,1,2}) for(bool reversed : {false,true}) for(uint32_t eyeIndex : {0u,1u}) {
const bool occluded=coverage==1;
gfx::StereoReplayFrame frame{};
frame.cockpit.unitsPerMeter=100;
frame.cockpit.active=true;frame.cockpit.wheelAngle=0.35f;
frame.cockpit.nativeWheel=original;
frame.cockpit.bike=bike;frame.cockpit.handlebarRadius=0.25f;
const float handlePose[12]{1,0,0,0, 0,0,1,-0.3f, 0,-1,0,-0.42f};
std::memcpy(frame.cockpit.seatFromHandlebar,handlePose,sizeof(handlePose));
for(int hand=0;hand<2;++hand) {
auto& h=frame.cockpit.hands[hand];h.tracked=true;h.held=true;h.squeeze=1;
auto pose=gfx::cockpit::identity();pose[3]=hand?0.18f:-0.18f;pose[7]=-0.30f;pose[11]=-0.42f;
std::memcpy(h.seatFromGrip,pose.data(),sizeof(h.seatFromGrip));
}
wgpu::TextureDescriptor td{.usage=wgpu::TextureUsage::RenderAttachment|wgpu::TextureUsage::CopySrc,
.size={512,512,1},.format=wgpu::TextureFormat::RGBA8Unorm,.sampleCount=1};
auto output=g_device.CreateTexture(&td);
td.sampleCount=samples;td.usage=wgpu::TextureUsage::RenderAttachment;
auto color=g_device.CreateTexture(&td);
td.format=wgpu::TextureFormat::Depth24PlusStencil8;auto depth=g_device.CreateTexture(&td);
auto& eye=frame.eyes[eyeIndex];eye.target.colorView=samples==1?output.CreateView():color.CreateView();
if(samples>1) eye.target.resolveView=output.CreateView();
eye.target.depthFormat=td.format;eye.target.depthView=depth.CreateView();eye.target.size={512,512,1};eye.target.msaaSamples=samples;
eye.projection.m0[0]=1;eye.projection.m1[1]=1;
eye.projection.m0[2]=eyeIndex?0.06f:-0.06f;
auto view=gfx::cockpit::identity();view[7]=0.20f;
std::memcpy(frame.cockpit.eyeFromSeat[eyeIndex],view.data(),sizeof(frame.cockpit.eyeFromSeat[eyeIndex]));
auto encoder=g_device.CreateCommandEncoder();
const wgpu::RenderPassColorAttachment clear{.view=eye.target.colorView,.resolveTarget=eye.target.resolveView,
.loadOp=wgpu::LoadOp::Clear,.storeOp=wgpu::StoreOp::Store,.clearValue={0.06,0.09,0.13,1}};
const wgpu::RenderPassDepthStencilAttachment sceneDepth{.view=eye.target.depthView,
.depthLoadOp=wgpu::LoadOp::Clear,.depthStoreOp=wgpu::StoreOp::Store,.depthClearValue=reversed?(occluded?0.8f:0.0f):(occluded?0.2f:1.0f),
.stencilLoadOp=wgpu::LoadOp::Clear,.stencilStoreOp=wgpu::StoreOp::Store,.stencilClearValue=0};
const wgpu::RenderPassDescriptor pd{.colorAttachmentCount=1,.colorAttachments=&clear,.depthStencilAttachment=&sceneDepth};
auto pass=encoder.BeginRenderPass(&pd);
if(coverage==2) {
wgpu::ShaderSourceWGSL code{};
code.code=R"(
@vertex fn vs(@builtin(vertex_index) i:u32) -> @builtin(position) vec4f {
let p=array<vec2f,6>(vec2f(0,-1),vec2f(1,-1),vec2f(0,1),vec2f(0,1),vec2f(1,-1),vec2f(1,1));
return vec4f(p[i],0.5,1);
}
@fragment fn fs() -> @location(0) vec4f { return vec4f(0.06,0.09,0.13,1); }
)";
wgpu::ShaderModuleDescriptor md{};md.nextInChain=&code;
auto shader=g_device.CreateShaderModule(&md);
const wgpu::ColorTargetState colorState{.format=wgpu::TextureFormat::RGBA8Unorm};
const wgpu::FragmentState fragment{.module=shader,.entryPoint="fs",.targetCount=1,.targets=&colorState};
const wgpu::DepthStencilState ds{.format=eye.target.depthFormat,.depthWriteEnabled=true,.depthCompare=wgpu::CompareFunction::Always};
wgpu::RenderPipelineDescriptor desc{};desc.vertex={.module=shader,.entryPoint="vs"};
desc.fragment=&fragment;desc.depthStencil=&ds;desc.multisample.count=samples;
auto wall=g_device.CreateRenderPipeline(&desc);pass.SetPipeline(wall);pass.Draw(6);
}
if(coverage==2) gfx::cockpit::render(encoder,frame,eyeIndex,reversed?gfx::cockpit::SceneDepth{0,2,true}:gfx::cockpit::SceneDepth{-1,-2,true},&pass);
pass.End();
if(coverage!=2) gfx::cockpit::render(encoder,frame,eyeIndex,reversed?gfx::cockpit::SceneDepth{0,2,true}:gfx::cockpit::SceneDepth{-1,-2,true});
if(hud) {
// An opaque black screen and coloured HUD with depth testing disabled used
// to overwrite the hands. Exercise a later pass too: the mask must survive.
const wgpu::RenderPassColorAttachment load{.view=eye.target.colorView,.resolveTarget=eye.target.resolveView,
.loadOp=wgpu::LoadOp::Load,.storeOp=wgpu::StoreOp::Store};
const wgpu::RenderPassDepthStencilAttachment loadDepth{.view=eye.target.depthView,
.depthLoadOp=wgpu::LoadOp::Load,.depthStoreOp=wgpu::StoreOp::Store,
.stencilLoadOp=wgpu::LoadOp::Load,.stencilStoreOp=wgpu::StoreOp::Store};
const wgpu::RenderPassDescriptor hudPass{.colorAttachmentCount=1,.colorAttachments=&load,.depthStencilAttachment=&loadDepth};
auto overlay=encoder.BeginRenderPass(&hudPass);
wgpu::ShaderSourceWGSL code{};
code.code=R"(
@vertex fn vs(@builtin(vertex_index) i:u32) -> @builtin(position) vec4f {
let p=array<vec2f,3>(vec2f(-1,-1),vec2f(3,-1),vec2f(-1,3));
return vec4f(p[i],0.5,1);
}
@fragment fn fs(@builtin(position) p:vec4f) -> @location(0) vec4f {
return select(vec4f(0,0,0,1),vec4f(1,0,0,1),p.x<256);
}
)";
wgpu::ShaderModuleDescriptor md{};md.nextInChain=&code;auto shader=g_device.CreateShaderModule(&md);
const wgpu::ColorTargetState colorState{.format=wgpu::TextureFormat::RGBA8Unorm};
const wgpu::FragmentState fragment{.module=shader,.entryPoint="fs",.targetCount=1,.targets=&colorState};
const wgpu::StencilFaceState mask{.compare=wgpu::CompareFunction::Equal};
const wgpu::DepthStencilState ds{.format=eye.target.depthFormat,.depthWriteEnabled=true,
.depthCompare=wgpu::CompareFunction::Always,.stencilFront=mask,.stencilBack=mask,.stencilReadMask=1,.stencilWriteMask=0};
wgpu::RenderPipelineDescriptor desc{};desc.vertex={.module=shader,.entryPoint="vs"};
desc.fragment=&fragment;desc.depthStencil=&ds;desc.multisample.count=samples;
auto screen=g_device.CreateRenderPipeline(&desc);overlay.SetPipeline(screen);overlay.Draw(3);overlay.End();
}
const wgpu::BufferDescriptor bd{.usage=wgpu::BufferUsage::CopyDst|wgpu::BufferUsage::MapRead,.size=512*512*4};
auto readback=g_device.CreateBuffer(&bd);
const wgpu::TexelCopyTextureInfo src{.texture=output};
const wgpu::TexelCopyBufferInfo dst{.layout={.bytesPerRow=2048,.rowsPerImage=512},.buffer=readback};
const wgpu::Extent3D extent{512,512,1};encoder.CopyTextureToBuffer(&src,&dst,&extent);
auto commands=encoder.Finish();g_device.GetQueue().Submit(1,&commands);
bool mapped=false;
future=readback.MapAsync(wgpu::MapMode::Read,0,512*512*4,wgpu::CallbackMode::WaitAnyOnly,
[&](wgpu::MapAsyncStatus status,wgpu::StringView) { mapped=status==wgpu::MapAsyncStatus::Success; });
if(instance.WaitAny(future,5000000000)!=wgpu::WaitStatus::Success||!mapped) return 1;
const auto* bytes=static_cast<const unsigned char*>(readback.GetConstMappedRange());
size_t bright=0;
for(size_t i=0;i<512*512;++i) if(bytes[4*i]>90&&bytes[4*i+1]>90&&bytes[4*i+2]>90) ++bright;
if(hud && !(bytes[0]>250 && bytes[1]==0 && bytes[2]==0)) {
std::cerr<<"HUD missing outside cockpit mask\n";++errors;
}
static size_t expectedBright[3][2][2]{};
auto& expected=expectedBright[coverage][reversed][eyeIndex];
if(!hud) expected=bright;
else if(bright+64<expected || bright>expected+64) { std::cerr<<"HUD changed visible cockpit pixels\n";++errors; }
if(occluded ? bright!=0 : bright<1000) { std::cerr<<"Incorrect hands/wheel occlusion\n";++errors; }
if(coverage==2) {
size_t left=0,right=0;
for(size_t y=0;y<512;++y) for(size_t x=0;x<512;++x) {
const auto i=y*512+x;
if(bytes[4*i]>90&&bytes[4*i+1]>90&&bytes[4*i+2]>90) (x<256?left:right)++;
}
if(left<500||right>8) { std::cerr<<"Partial wall occlusion failed for eye "<<eyeIndex<<'\n';++errors; }
}
if(samples==4 && !original && !occluded) {
std::ofstream image("cockpit-preview.ppm",std::ios::binary);image<<"P6\n512 512\n255\n";
for(size_t i=0;i<512*512;++i) image.write(reinterpret_cast<const char*>(bytes+i*4),3);
}
readback.Unmap();
std::cout<<(bike?"Bike ":"Kart ")<<(original?"native hands: ":"VR controls: ")<<samples<<"x MSAA: "<<bright<<" visible geometry pixels\n";
}
gfx::cockpit::shutdown();g_queue=nullptr;g_device.Destroy();g_device=nullptr;
return errors?1:0;
}
+207
View File
@@ -0,0 +1,207 @@
// SPDX-License-Identifier: GPL-3.0-or-later
// Native steering wheel: indexed-matrix ownership and array matching. The
// ownership cases come from heurazy's mario-kart-wii-VR-port test set.
#include <gtest/gtest.h>
#include <array>
#include <vector>
#include "gx_test_common.hpp"
#include "gx/native_wheel.hpp"
namespace {
// Three positions, 12 bytes each. Position 1 is on the wheel; 0 and 2 belong to
// the body/wing.
struct OwnershipFixture {
std::array<uint8_t, 36> original{};
std::array<uint8_t, 36> replacement{};
// PN matrix byte, texture matrix byte, big-endian position index.
std::array<uint8_t, 12> vertices{0, 30, 0, 0, 0, 30, 0, 1, 3, 30, 0, 2};
OwnershipFixture() { replacement[12] = 7; }
bool matches(uint16_t mask) const {
return aurora::NativeWheelDrawMatches(original, replacement, 12, vertices, 4, 2, 2, mask);
}
};
TEST(NativeWheelMatch, MixedBodyAndWingDrawAcceptsWheelPositionsOnLocalBody) {
OwnershipFixture f;
EXPECT_TRUE(f.matches(1));
}
TEST(NativeWheelMatch, UnrelatedWingJointCannotAnimateWheel) {
OwnershipFixture f;
EXPECT_FALSE(f.matches(2));
}
TEST(NativeWheelMatch, OpponentWithoutLocalMatrixIsRejected) {
OwnershipFixture f;
EXPECT_FALSE(f.matches(0));
}
TEST(NativeWheelMatch, WheelPositionUnderAnotherMatrixIsRejected) {
OwnershipFixture f;
f.vertices[4] = 6;
EXPECT_FALSE(f.matches(1));
EXPECT_TRUE(f.matches(4)) << "local body can occupy a different palette slot";
}
TEST(NativeWheelMatch, MalformedMatrixSelectorRejected) {
OwnershipFixture f;
f.vertices[4] = 1;
EXPECT_FALSE(f.matches(1));
}
TEST(NativeWheelMatch, OutOfRangePositionIndexRejected) {
OwnershipFixture f;
f.vertices[11] = 3;
EXPECT_FALSE(f.matches(1));
}
TEST(NativeWheelMatch, SelectionTouchingAnotherJointCannotDeformIt) {
OwnershipFixture f;
f.replacement[24] = 1;
EXPECT_FALSE(f.matches(1));
}
TEST(NativeWheelMatch, EightBitIndicesAndLayoutValidation) {
OwnershipFixture f;
std::array<uint8_t, 4> indexed8{0, 1, 3, 2};
EXPECT_TRUE(aurora::NativeWheelDrawMatches(f.original, f.replacement, 12, indexed8, 2, 1, 1, 1));
EXPECT_FALSE(aurora::NativeWheelDrawMatches(f.original, f.replacement, 12, indexed8, 2, 2, 1, 1))
<< "invalid vertex layout rejected";
EXPECT_FALSE(aurora::NativeWheelDrawMatches(f.original, f.original, 12, indexed8, 2, 1, 1, 1))
<< "unchanged positions are not counted as animation";
}
// Draw-time matching against the GX position-matrix palette.
class NativeWheelArrayTest : public ::testing::Test {
protected:
void SetUp() override {
aurora::gx::g_gxState = {};
aurora::gx::nativeWheelArrays.clear();
aurora::gx::nativeWheelMatches = 0;
aurora::gx::NativeWheelArray replacement;
replacement.source = source.data();
replacement.bytes.assign(36, 0);
replacement.bytes[12] = 1;
aurora::gx::nativeWheelArrays.push_back(replacement);
array.data = source.data();
array.size = 36;
array.stride = 12;
}
void TearDown() override {
aurora::gx::g_gxState = {};
aurora::gx::nativeWheelArrays.clear();
}
static float* pos(uint32_t slot) { return reinterpret_cast<float*>(&aurora::gx::g_gxState.pnMtx[slot].pos); }
std::array<uint8_t, 36> source{};
aurora::gx::AttrArray array{};
};
TEST_F(NativeWheelArrayTest, MatchesLocalVehicleMatrixOnly) {
EXPECT_NE(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr);
EXPECT_EQ(aurora::gx::nativeWheelMatches, 1u);
// Same asset drawn by an opponent: a translated matrix.
pos(0)[3] = 100;
EXPECT_EQ(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr);
EXPECT_EQ(aurora::gx::nativeWheelMatches, 1u);
}
TEST_F(NativeWheelArrayTest, UnboundSourceIsNotAWheelArray) {
std::array<uint8_t, 36> other{};
aurora::gx::AttrArray unrelated = array;
unrelated.data = other.data();
EXPECT_EQ(aurora::gx::native_wheel_array(unrelated, nullptr, 0, 0, 0), nullptr);
EXPECT_TRUE(aurora::gx::native_wheel_source(source.data()));
EXPECT_FALSE(aurora::gx::native_wheel_source(other.data()));
}
TEST_F(NativeWheelArrayTest, MultiJointBodyChecksPerPositionOwnership) {
// Multi-joint body with an independently animated wing, as in Mario's
// mb/mc kart models. Only the wheel position uses the matching body slot.
aurora::gx::g_gxState.vtxDesc[GX_VA_PNMTXIDX] = GX_DIRECT;
aurora::gx::g_gxState.vtxDesc[GX_VA_POS] = GX_INDEX16;
for (uint32_t slot = 0; slot < aurora::gx::MaxPnMtx; ++slot) pos(slot)[3] = 100;
pos(2)[3] = 0;
uint8_t vertices[]{6, 0, 1, 3, 0, 2};
EXPECT_NE(aurora::gx::native_wheel_array(array, vertices, sizeof(vertices), 3, 1), nullptr);
vertices[0] = 0; // opponent wheel while an unrelated palette slot still matches
EXPECT_EQ(aurora::gx::native_wheel_array(array, vertices, sizeof(vertices), 3, 1), nullptr);
}
TEST_F(NativeWheelArrayTest, NonFiniteMatrixNeverMatches) {
pos(0)[0] = __builtin_nanf("");
EXPECT_EQ(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr);
}
// Draw merging around an animated array. A merged draw appends its vertices to the previous draw's range and renders
// through that draw's array binding, so two draws may merge only when they resolved the same replacement. The kart's
// display list is hundreds of same-state primitives, and merging them is worth several ms an eye on a tiler.
class NativeWheelMergeTest : public GXFifoTest {
protected:
void SetUp() override {
GXFifoTest::SetUp();
aurora::gfx::testing::use_draw_command_tracking(true);
aurora::gx::nativeWheelArrays.clear();
aurora::gx::nativeWheelLastDecision = nullptr;
aurora::gx::nativeWheelLastDrawCommand = nullptr;
aurora::gx::NativeWheelArray replacement;
replacement.source = source.data();
replacement.bytes.assign(source.size(), 0);
replacement.bytes[12] = 1; // one animated position
aurora::gx::nativeWheelArrays.push_back(replacement);
auto& state = aurora::gx::g_gxState;
state.lastVtxFmt = GX_VTXFMT0;
state.lastVtxSize = 1;
state.vtxDesc[GX_VA_POS] = GX_INDEX8;
state.arrays[GX_VA_POS].data = source.data();
state.arrays[GX_VA_POS].size = static_cast<u32>(source.size());
state.arrays[GX_VA_POS].stride = 12;
state.stateDirty = true;
}
void TearDown() override {
aurora::gx::nativeWheelArrays.clear();
aurora::gx::nativeWheelLastDecision = nullptr;
aurora::gx::nativeWheelLastDrawCommand = nullptr;
}
void draw() {
std::vector<u8> fifo{static_cast<u8>(GX_TRIANGLES) | static_cast<u8>(GX_VTXFMT0), 0, 3, 0, 1, 2};
decode_fifo(fifo);
}
// The local vehicle's matrix: the replacement's model-view is all zeroes, and so is a default palette slot.
void makeOpponent() { reinterpret_cast<float*>(&aurora::gx::g_gxState.pnMtx[0].pos)[3] = 100.f; }
std::array<uint8_t, 36> source{};
};
TEST_F(NativeWheelMergeTest, PrimitivesSharingTheAnimatedArrayStillMerge) {
draw();
ASSERT_EQ(aurora::gx::nativeWheelLastDecision, &aurora::gx::nativeWheelArrays.front());
draw();
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u);
}
TEST_F(NativeWheelMergeTest, PrimitivesThatTakeTheOriginalArrayAlsoStillMerge) {
makeOpponent();
draw();
ASSERT_EQ(aurora::gx::nativeWheelLastDecision, nullptr);
draw();
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u);
}
TEST_F(NativeWheelMergeTest, OpponentPrimitiveNeverFoldsIntoAnAnimatedDraw) {
draw();
makeOpponent();
draw();
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u) << "the merged whole would render the opponent animated";
}
TEST_F(NativeWheelMergeTest, AnimatedPrimitiveNeverFoldsIntoAnOpponentDraw) {
makeOpponent();
draw();
reinterpret_cast<float*>(&aurora::gx::g_gxState.pnMtx[0].pos)[3] = 0.f;
draw();
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u) << "the wheel would render on the original vertices";
}
} // namespace
@@ -0,0 +1,162 @@
// Opt-in real GPU test for the custom Dawn DLL. No headset required.
// Exercises the MinGW/MSVC C ABI, borrowed-image lifetime and repeated layout
// transitions. Run with VK_INSTANCE_LAYERS=VK_LAYER_KHRONOS_validation as well.
#define NOMINMAX
#include <windows.h>
#include <vulkan/vulkan.h>
#include <dawn/webgpu_cpp.h>
#include <aurora/dawn_vulkan_abi.h>
#include <array>
#include <cstdio>
#include <cstdlib>
static void Check(bool ok) { if (!ok) { std::fputs("Native Vulkan bridge smoke failed\n", stderr); std::abort(); } }
static PFN_vkGetInstanceProcAddr hookProc;
static VkInstance hookInstance;
static int instances = 0, devices = 0, selections = 0;
static int32_t CreateInstance(void*, void* proc, const void* info, const void* allocator, void** out) {
++instances; hookProc = reinterpret_cast<PFN_vkGetInstanceProcAddr>(proc);
auto create = reinterpret_cast<PFN_vkCreateInstance>(hookProc(nullptr, "vkCreateInstance"));
auto result = create(static_cast<const VkInstanceCreateInfo*>(info), static_cast<const VkAllocationCallbacks*>(allocator), &hookInstance);
*out = hookInstance; return result;
}
static int32_t CreateDevice(void*, void*, void* physical, const void* info, const void* allocator, void** out) {
++devices;
auto create = reinterpret_cast<PFN_vkCreateDevice>(hookProc(hookInstance, "vkCreateDevice"));
return create(static_cast<VkPhysicalDevice>(physical), static_cast<const VkDeviceCreateInfo*>(info),
static_cast<const VkAllocationCallbacks*>(allocator), reinterpret_cast<VkDevice*>(out));
}
static int32_t SelectPhysical(void*, void* instance, void** out) {
++selections;
auto enumerate = reinterpret_cast<PFN_vkEnumeratePhysicalDevices>(hookProc(static_cast<VkInstance>(instance), "vkEnumeratePhysicalDevices"));
uint32_t count = 1;
auto result = enumerate(static_cast<VkInstance>(instance), &count, reinterpret_cast<VkPhysicalDevice*>(out));
return result == VK_INCOMPLETE ? VK_SUCCESS : result;
}
int main() {
auto dll = LoadLibraryW(L"webgpu_dawn.dll"); Check(dll != nullptr);
#define API(name, type) auto name = reinterpret_cast<type>(GetProcAddress(dll, "AuroraDawnVulkan" #name)); Check(name != nullptr)
API(Version, AuroraDawnVulkanVersionFn);
API(Configure, AuroraDawnVulkanConfigureFn);
API(GetHandles, AuroraDawnVulkanHandlesFn);
API(Wrap, AuroraDawnVulkanWrapFn);
API(Release, AuroraDawnVulkanReleaseFn);
API(Lock, AuroraDawnVulkanLockFn);
API(Unlock, AuroraDawnVulkanUnlockFn);
API(Drain, AuroraDawnVulkanDrainFn);
Check(Version() == AURORA_DAWN_VULKAN_ABI);
AuroraDawnVulkanHooks hooks{nullptr, CreateInstance, CreateDevice, SelectPhysical};
Check(Configure(&hooks));
wgpu::InstanceFeatureName feature = wgpu::InstanceFeatureName::TimedWaitAny;
wgpu::InstanceDescriptor instanceDesc;
instanceDesc.requiredFeatureCount = 1; instanceDesc.requiredFeatures = &feature;
std::fputs("Creating WebGPU instance\n", stderr);
auto instance = wgpu::CreateInstance(&instanceDesc);
wgpu::RequestAdapterOptions options; options.backendType = wgpu::BackendType::Vulkan;
wgpu::Adapter adapter;
std::fputs("Requesting Vulkan adapter\n", stderr);
auto future = instance.RequestAdapter(&options, wgpu::CallbackMode::WaitAnyOnly,
[&](wgpu::RequestAdapterStatus status, wgpu::Adapter value, wgpu::StringView) {
Check(status == wgpu::RequestAdapterStatus::Success); adapter = std::move(value);
});
Check(instance.WaitAny(future, 10'000'000'000) == wgpu::WaitStatus::Success);
wgpu::FeatureName sync = wgpu::FeatureName::ImplicitDeviceSynchronization;
wgpu::DeviceDescriptor deviceDesc;
deviceDesc.requiredFeatureCount = 1; deviceDesc.requiredFeatures = &sync;
deviceDesc.SetUncapturedErrorCallback([](const wgpu::Device&, wgpu::ErrorType, wgpu::StringView message) {
std::fprintf(stderr, "Dawn: %.*s\n", static_cast<int>(message.length), message.data); Check(false);
});
std::fputs("Creating Vulkan device\n", stderr);
auto device = adapter.CreateDevice(&deviceDesc); Check(!!device);
Check(instances > 0 && devices > 0 && selections > 0);
std::fputs("Getting native device\n", stderr);
AuroraDawnVulkanHandles handles{}; Check(GetHandles(device.Get(), &handles));
VkDevice vkDevice = static_cast<VkDevice>(handles.device);
VkPhysicalDevice physical = static_cast<VkPhysicalDevice>(handles.physicalDevice);
auto loader = LoadLibraryW(L"vulkan-1.dll"); Check(loader != nullptr);
auto getProc = reinterpret_cast<PFN_vkGetInstanceProcAddr>(GetProcAddress(loader, "vkGetInstanceProcAddr"));
#define VK(name) auto name = reinterpret_cast<PFN_##name>(getProc(static_cast<VkInstance>(handles.instance), #name)); Check(name != nullptr)
VK(vkCreateImage); VK(vkGetImageMemoryRequirements); VK(vkGetPhysicalDeviceMemoryProperties);
VK(vkAllocateMemory); VK(vkBindImageMemory); VK(vkCreateCommandPool); VK(vkAllocateCommandBuffers);
VK(vkBeginCommandBuffer); VK(vkCmdPipelineBarrier); VK(vkEndCommandBuffer); VK(vkGetDeviceQueue);
VK(vkQueueSubmit); VK(vkQueueWaitIdle); VK(vkDestroyCommandPool); VK(vkDestroyImage); VK(vkFreeMemory);
VkImageCreateInfo imageInfo{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
imageInfo.imageType = VK_IMAGE_TYPE_2D; imageInfo.format = VK_FORMAT_R8G8B8A8_UNORM;
imageInfo.extent = {64, 16, 1}; imageInfo.mipLevels = 1; imageInfo.arrayLayers = 1;
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT; imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
imageInfo.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
std::fputs("Creating borrowed Vulkan image\n", stderr);
VkImage image; Check(vkCreateImage(vkDevice, &imageInfo, nullptr, &image) == VK_SUCCESS);
VkMemoryRequirements requirements; vkGetImageMemoryRequirements(vkDevice, image, &requirements);
VkPhysicalDeviceMemoryProperties properties; vkGetPhysicalDeviceMemoryProperties(physical, &properties);
uint32_t memoryType = 0;
while (!(requirements.memoryTypeBits & (1u << memoryType))) ++memoryType;
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
allocation.allocationSize = requirements.size; allocation.memoryTypeIndex = memoryType;
VkDeviceMemory memory; Check(vkAllocateMemory(vkDevice, &allocation, nullptr, &memory) == VK_SUCCESS);
Check(vkBindImageMemory(vkDevice, image, memory, 0) == VK_SUCCESS);
VkCommandPoolCreateInfo poolInfo{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; poolInfo.queueFamilyIndex = handles.queueFamily;
VkCommandPool pool; Check(vkCreateCommandPool(vkDevice, &poolInfo, nullptr, &pool) == VK_SUCCESS);
VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
alloc.commandPool = pool; alloc.commandBufferCount = 1; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
VkCommandBuffer commands; Check(vkAllocateCommandBuffers(vkDevice, &alloc, &commands) == VK_SUCCESS);
VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
Check(vkBeginCommandBuffer(commands, &begin) == VK_SUCCESS);
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
barrier.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; barrier.newLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL;
barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
barrier.image = image; barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
barrier.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
vkCmdPipelineBarrier(commands, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT,
0, 0, nullptr, 0, nullptr, 1, &barrier);
Check(vkEndCommandBuffer(commands) == VK_SUCCESS);
VkQueue queue; vkGetDeviceQueue(vkDevice, handles.queueFamily, handles.queueIndex, &queue);
VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; submit.commandBufferCount = 1; submit.pCommandBuffers = &commands;
std::fputs("Locking native queue\n", stderr);
auto guard = Lock(device.Get());
Check(vkQueueSubmit(queue, 1, &submit, VK_NULL_HANDLE) == VK_SUCCESS);
Check(vkQueueWaitIdle(queue) == VK_SUCCESS); Unlock(guard);
wgpu::TextureDescriptor textureDesc;
textureDesc.size = {64, 16, 1}; textureDesc.format = wgpu::TextureFormat::RGBA8Unorm;
textureDesc.usage = wgpu::TextureUsage::CopyDst | wgpu::TextureUsage::CopySrc | wgpu::TextureUsage::RenderAttachment;
std::fputs("Wrapping borrowed image\n", stderr);
auto borrowed = wgpu::Texture::Acquire(static_cast<WGPUTexture>(Wrap(device.Get(), &textureDesc, reinterpret_cast<uint64_t>(image))));
Check(!!borrowed);
std::fputs("Creating copy resources\n", stderr);
auto source = device.CreateTexture(&textureDesc);
wgpu::BufferDescriptor bufferDesc; bufferDesc.size = 4096;
bufferDesc.usage = wgpu::BufferUsage::CopyDst | wgpu::BufferUsage::MapRead;
auto readback = device.CreateBuffer(&bufferDesc);
std::fputs("Running GPU copy\n", stderr);
for (uint8_t value : {17, 99, 201}) {
std::array<uint8_t, 4096> pixels; pixels.fill(value);
wgpu::TexelCopyTextureInfo sourceInfo; sourceInfo.texture = source;
wgpu::TexelCopyTextureInfo targetInfo; targetInfo.texture = borrowed;
wgpu::TexelCopyBufferLayout layout; layout.bytesPerRow = 256; layout.rowsPerImage = 16;
wgpu::Extent3D extent{64, 16, 1};
device.GetQueue().WriteTexture(&sourceInfo, pixels.data(), pixels.size(), &layout, &extent);
std::fputs("WriteTexture done\n", stderr);
auto encoder = device.CreateCommandEncoder(); encoder.CopyTextureToTexture(&sourceInfo, &targetInfo, &extent);
auto copy = encoder.Finish(); device.GetQueue().Submit(1, &copy);
std::fputs("Copy submitted\n", stderr);
void* textures[] = {borrowed.Get()}; Check(Release(device.Get(), textures, 1));
std::fputs("Release transition done\n", stderr);
encoder = device.CreateCommandEncoder();
wgpu::TexelCopyBufferInfo destination; destination.buffer = readback; destination.layout = layout;
encoder.CopyTextureToBuffer(&targetInfo, &destination, &extent);
copy = encoder.Finish(); device.GetQueue().Submit(1, &copy);
Check(Release(device.Get(), textures, 1));
auto mapped = readback.MapAsync(wgpu::MapMode::Read, 0, 4096, wgpu::CallbackMode::WaitAnyOnly,
[](wgpu::MapAsyncStatus status, wgpu::StringView) { Check(status == wgpu::MapAsyncStatus::Success); });
Check(instance.WaitAny(mapped, 10'000'000'000) == wgpu::WaitStatus::Success);
const auto* bytes = static_cast<const uint8_t*>(readback.GetConstMappedRange());
for (size_t i = 0; i < pixels.size(); ++i) Check(bytes[i] == value);
readback.Unmap();
}
Check(Drain(device.Get())); borrowed = nullptr;
// The wrapper must not free the runtime-owned image or its memory.
vkDestroyImage(vkDevice, image, nullptr); vkFreeMemory(vkDevice, memory, nullptr);
vkDestroyCommandPool(vkDevice, pool, nullptr);
Check(Configure(nullptr));
std::puts("Native Vulkan bridge: three GPU copy/readback cycles passed");
}