mirror of
https://github.com/mitch030504/Wiicompiled_VR_Frame.git
synced 2026-10-06 08:00:25 +02:00
Merge remote-tracking branch 'upstream/openxr-work' into codex/standalone-selective-integration
# Conflicts: # docs/quest-port.md
This commit is contained in:
commit
0e52cfb726
141 files changed
+14235
-1090
No files matched your search
@@ -38,7 +38,7 @@ if (AURORA_ENABLE_GX)
|
||||
# OpenXR-enabled runtime can fail back to desktop mode at runtime instead
|
||||
# of producing unresolved interop references.
|
||||
if (CMAKE_SYSTEM_NAME STREQUAL Windows)
|
||||
target_sources(aurora_core PRIVATE lib/webgpu/d3d12_interop.cpp)
|
||||
target_sources(aurora_core PRIVATE lib/webgpu/d3d12_interop.cpp lib/webgpu/vulkan_win32_interop.cpp)
|
||||
endif ()
|
||||
# Android/Vulkan counterpart: the AHardwareBuffer stereo bridge. The file
|
||||
# compiles to C ABI stubs on every other platform so the runtime's OpenXR
|
||||
|
||||
@@ -116,6 +116,54 @@ typedef enum {
|
||||
// accepted through aurora_end_frame_tagged() with an exact matching tag.
|
||||
#define AURORA_STEREO_CONTENT_TAG_UNKNOWN UINT64_MAX
|
||||
|
||||
/**
|
||||
* VR cockpit overlay: tracked hands and, when the vehicle's own wheel cannot be
|
||||
* animated, a synthetic steering wheel or handlebar. Everything is in metres
|
||||
* in a seated frame (+X right, +Y up, -Z forward) whose origin is the headset's
|
||||
* immersive base position. Aurora draws it per eye after the scene, depth-tested
|
||||
* against the scene with the scene's own depth mapping.
|
||||
*/
|
||||
typedef struct {
|
||||
bool tracked;
|
||||
bool held;
|
||||
float squeeze;
|
||||
float seatFromGrip[12];
|
||||
} AuroraCockpitHand;
|
||||
|
||||
typedef struct {
|
||||
bool active;
|
||||
float wheelAngle;
|
||||
// The vehicle's own wheel is animated in the scene, so no synthetic wheel is drawn.
|
||||
bool nativeWheel;
|
||||
bool bike;
|
||||
float handlebarRadius;
|
||||
// World units per metre used to build this packet's eye transforms.
|
||||
float unitsPerMeter;
|
||||
float seatFromHandlebar[12];
|
||||
float eyeFromSeat[AURORA_STEREO_EYE_COUNT][12];
|
||||
AuroraCockpitHand hands[2];
|
||||
} AuroraCockpit;
|
||||
|
||||
typedef struct {
|
||||
float position[3];
|
||||
int16_t joints[4];
|
||||
float weights[4];
|
||||
} AuroraVRHandVertex;
|
||||
|
||||
/**
|
||||
* Shows the headset settings panel (aurora_imgui_set_stereo_overlay) as the
|
||||
* OpenXR backend's own compositor quad layer instead of drawing it into the
|
||||
* eyes, so the eye resolution no longer limits its text. The backend then asks
|
||||
* for the panel image as an extra stereo target. Any thread.
|
||||
*/
|
||||
void aurora_set_stereo_panel_layer(bool enabled);
|
||||
|
||||
// Copies optional runtime-provided hand meshes (XR_FB_hand_tracking_mesh, 26
|
||||
// joints). Null clears to the procedural glove. Bind poses: x,y,z,w,px,py,pz.
|
||||
void aurora_set_vr_hand_mesh(uint32_t hand, const AuroraVRHandVertex* vertices, uint32_t vertexCount,
|
||||
const uint16_t* indices, uint32_t indexCount, const float* bindPoses,
|
||||
const int32_t* parents, uint32_t jointCount);
|
||||
|
||||
/**
|
||||
* Stereo data for one sealed GX frame. frameToken is opaque to Aurora and is
|
||||
* forwarded unchanged to the internal stereo output sink. contentTag must
|
||||
@@ -130,6 +178,8 @@ typedef struct {
|
||||
// Predicted display time converted to std::chrono::steady_clock nanoseconds.
|
||||
// Zero disables temporal interpolation for this packet.
|
||||
uint64_t displayTimeNanos;
|
||||
// Optional; inactive when zero-initialised.
|
||||
AuroraCockpit cockpit;
|
||||
} AuroraStereoFrame;
|
||||
|
||||
/**
|
||||
@@ -218,6 +268,17 @@ void aurora_end_frame();
|
||||
// Seal the current frame with an opaque application safety tag. Aurora rejects
|
||||
// an immersive provider packet unless its contentTag matches this exact frame.
|
||||
void aurora_end_frame_tagged(uint64_t contentTag);
|
||||
// aurora_end_frame_tagged() plus the host-owned ImGui frame to present with it (the handle from
|
||||
// aurora_imgui_host_frame_end(), which this call consumes; NULL presents no host ImGui frame).
|
||||
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame);
|
||||
// When the host pumps SDL events itself (aurora_update() on the window's thread) and drives
|
||||
// begin/end frame from another thread, this stops those calls from pumping events.
|
||||
void aurora_set_host_event_pump(bool hostPumps);
|
||||
typedef void (*AuroraFrameLogCallback)(char* buffer, uint32_t bufferSize, double windowSeconds,
|
||||
uint32_t frames);
|
||||
// Called with each five-second frame-rate log window (where that log is enabled); a non-empty
|
||||
// buffer is logged as one extra line.
|
||||
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback);
|
||||
/**
|
||||
* Relocates the immersive camera for the frame about to be sealed.
|
||||
*
|
||||
@@ -236,10 +297,28 @@ void aurora_end_frame_tagged(uint64_t contentTag);
|
||||
* provider, which cannot know which frame will consume its packet.
|
||||
*/
|
||||
void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]);
|
||||
// As above, also naming the world units per metre the anchor was built with.
|
||||
// The sealed frame then owns that scale: each eye's head/IPD translation is
|
||||
// rescaled from the packet's AuroraCockpit::unitsPerMeter to it.
|
||||
void aurora_set_stereo_scene_anchor_scaled(const float anchorFromScene[12], float unitsPerMeter);
|
||||
// Select Player 1's subview for immersive replay of 2-4 local screens.
|
||||
// Producer-thread, per-frame metadata, consumed by the next end_frame call.
|
||||
// One (the default) keeps full-frame replay. Desktop rendering is unaffected.
|
||||
void aurora_set_stereo_local_player_count(uint32_t count);
|
||||
/**
|
||||
* VR native steering wheel: replacement position arrays for the local vehicle.
|
||||
*
|
||||
* GX (command processor) thread only, in order with the frame's draws: a host
|
||||
* that runs GX on its own thread must post these there. `source` is the host
|
||||
* pointer the game binds with GXSetArray; `replacement` is copied (at most 64
|
||||
* KiB) and applies solely to draws that bind that array with a position matrix
|
||||
* equal to `modelView` (row-major 3x4), so shared opponent models stay intact.
|
||||
* Clearing reports how many draws the previous set matched.
|
||||
*/
|
||||
void aurora_clear_native_wheel_vertices(void);
|
||||
void aurora_set_native_wheel_vertices(const void* source, const void* replacement, uint32_t size,
|
||||
const float* modelView);
|
||||
uint32_t aurora_native_wheel_draw_count(void);
|
||||
typedef void (*AuroraFrameWorkerWaitCallback)();
|
||||
// Called from the producer thread at bounded intervals while Aurora waits for
|
||||
// the asynchronous frame worker. The callback must not enter Aurora.
|
||||
|
||||
@@ -64,6 +64,17 @@ bool aurora_d3d12_set_stereo_targets(uint64_t frameToken,
|
||||
const AuroraD3D12StereoTarget* targets,
|
||||
uint32_t targetCount);
|
||||
|
||||
/**
|
||||
* The same, plus the headset settings panel's quad-layer image when `panel` is
|
||||
* not null (aurora_set_stereo_panel_layer): Aurora copies the panel into it, or
|
||||
* a transparent image while the panel is not showing, with the eyes and under
|
||||
* the same completion callback.
|
||||
*/
|
||||
bool aurora_d3d12_set_stereo_targets_with_panel(uint64_t frameToken,
|
||||
const AuroraD3D12StereoTarget* targets,
|
||||
uint32_t targetCount,
|
||||
const AuroraD3D12StereoTarget* panel);
|
||||
|
||||
/**
|
||||
* Withdraws frameToken only while its target has not been encoded. This is
|
||||
* safe to race with Aurora's frame worker: false means the worker already owns
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
#pragma once
|
||||
#include <stdint.h>
|
||||
// Versioned C ABI: no STL objects or compiler-specific C++ symbols cross the
|
||||
// MSVC Dawn DLL / LLVM-MinGW runtime boundary. Vulkan handles are borrowed.
|
||||
#define AURORA_DAWN_VULKAN_ABI 1
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
typedef struct {
|
||||
void* userdata;
|
||||
int32_t (*createInstance)(void*, void* getProc, const void* info, const void* allocator, void** instance);
|
||||
int32_t (*createDevice)(void*, void* getProc, void* physical, const void* info, const void* allocator, void** device);
|
||||
int32_t (*getPhysicalDevice)(void*, void* instance, void** physical);
|
||||
} AuroraDawnVulkanHooks;
|
||||
typedef struct {
|
||||
void* instance;
|
||||
void* physicalDevice;
|
||||
void* device;
|
||||
uint32_t queueFamily;
|
||||
uint32_t queueIndex;
|
||||
} AuroraDawnVulkanHandles;
|
||||
typedef uint32_t (*AuroraDawnVulkanVersionFn)(void);
|
||||
typedef int (*AuroraDawnVulkanConfigureFn)(const AuroraDawnVulkanHooks*);
|
||||
typedef int (*AuroraDawnVulkanHandlesFn)(void* device, AuroraDawnVulkanHandles*);
|
||||
typedef void* (*AuroraDawnVulkanWrapFn)(void* device, const void* textureDescriptor, uint64_t image);
|
||||
typedef int (*AuroraDawnVulkanReleaseFn)(void* device, void* const* textures, uint32_t count);
|
||||
typedef void* (*AuroraDawnVulkanLockFn)(void* device);
|
||||
typedef void (*AuroraDawnVulkanUnlockFn)(void* guard);
|
||||
typedef int (*AuroraDawnVulkanDrainFn)(void* device);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
@@ -25,6 +25,14 @@ ImTextureID aurora_imgui_add_texture(uint32_t width, uint32_t height, const void
|
||||
// producer before the frame is sealed, and leave the draw data untouched until the frame worker is
|
||||
// done with that frame (aurora_wait_for_frame_worker). Null hides the panel.
|
||||
void aurora_imgui_set_stereo_overlay(ImDrawData* drawData, float widthFraction);
|
||||
// Host-owned ImGui frames for the desktop overlay. Begin starts the next frame on the calling
|
||||
// thread (which must be the window's thread, since the SDL backend reads the window there); end
|
||||
// renders it and returns a handle to a private copy of its draw data, which aurora_end_frame_ex()
|
||||
// consumes. A handle that is never presented is freed with aurora_imgui_host_frame_release().
|
||||
// From the first begin on, aurora no longer starts ImGui frames itself.
|
||||
void aurora_imgui_host_frame_begin(void);
|
||||
void* aurora_imgui_host_frame_end(void);
|
||||
void aurora_imgui_host_frame_release(void* imguiFrame);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <span>
|
||||
|
||||
namespace aurora {
|
||||
// Every referenced position that changes must belong to the local body matrix.
|
||||
// Unchanged positions may use other joints in the same indexed draw (wing/engine).
|
||||
inline bool NativeWheelDrawMatches(std::span<const uint8_t> original,
|
||||
std::span<const uint8_t> replacement, uint32_t positionStride,
|
||||
std::span<const uint8_t> vertices, uint32_t vertexStride, uint32_t positionOffset,
|
||||
uint32_t indexBytes, uint16_t matchingMatrices) {
|
||||
if(!positionStride || !vertexStride || (indexBytes!=1 && indexBytes!=2) ||
|
||||
positionOffset>=vertexStride || indexBytes>vertexStride-positionOffset ||
|
||||
original.size()!=replacement.size() || vertices.size()%vertexStride || !matchingMatrices) return false;
|
||||
bool changed=false;
|
||||
for(size_t start=0;start<vertices.size();start+=vertexStride) {
|
||||
const auto* vertex=vertices.data()+start;
|
||||
const uint32_t index=indexBytes==1 ? vertex[positionOffset]
|
||||
: (uint32_t(vertex[positionOffset])<<8)|vertex[positionOffset+1];
|
||||
const size_t offset=size_t(index)*positionStride;
|
||||
if(offset>original.size() || positionStride>original.size()-offset) return false;
|
||||
if(std::memcmp(original.data()+offset,replacement.data()+offset,positionStride)==0) continue;
|
||||
const uint32_t matrix=vertex[0]/3u;
|
||||
if(vertex[0]%3u || matrix>=16 || !(matchingMatrices&(1u<<matrix))) return false;
|
||||
changed=true;
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
}
|
||||
@@ -27,6 +27,8 @@ extern "C" {
|
||||
*/
|
||||
|
||||
enum { AURORA_VULKAN_STEREO_MAX_TARGETS = 2 };
|
||||
// The eyes, then the settings panel's quad-layer image when one was given.
|
||||
enum { AURORA_VULKAN_STEREO_MAX_RELEASES = AURORA_VULKAN_STEREO_MAX_TARGETS + 1 };
|
||||
|
||||
/**
|
||||
* Borrowed facts about Aurora's Dawn Vulkan device. colorVkFormat is the
|
||||
@@ -109,6 +111,18 @@ bool aurora_vulkan_set_stereo_targets(uint64_t frameToken,
|
||||
const AuroraVulkanStereoTarget* targets,
|
||||
uint32_t targetCount);
|
||||
|
||||
/**
|
||||
* The same, plus the headset settings panel's quad-layer buffer when `panel` is
|
||||
* not null (aurora_set_stereo_panel_layer): Aurora copies the panel into it, or
|
||||
* a transparent image while the panel is not showing, with the eyes. Its
|
||||
* release entry follows the eyes' in the submitted callback, whose
|
||||
* releaseCount then counts it too.
|
||||
*/
|
||||
bool aurora_vulkan_set_stereo_targets_with_panel(uint64_t frameToken,
|
||||
const AuroraVulkanStereoTarget* targets,
|
||||
uint32_t targetCount,
|
||||
const AuroraVulkanStereoTarget* panel);
|
||||
|
||||
/**
|
||||
* Withdraws frameToken only while its targets have not been encoded. Semantics
|
||||
* match aurora_d3d12_cancel_stereo_targets: false means the worker already
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
#pragma once
|
||||
#include <aurora/dawn_vulkan_abi.h>
|
||||
#include <aurora/d3d12_interop.h>
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
// Same pending-target/callback contract as D3D12. resource carries a VkImage
|
||||
// encoded as a pointer-sized value; colorDxgiFormat carries a VkFormat.
|
||||
bool aurora_vulkan_win32_configure(const AuroraDawnVulkanHooks* hooks);
|
||||
bool aurora_vulkan_win32_get_handles(AuroraDawnVulkanHandles* handles, int64_t* colorFormat);
|
||||
bool aurora_vulkan_win32_enable(AuroraD3D12StereoSubmittedCallback submitted, void* userdata);
|
||||
bool aurora_vulkan_win32_set_targets(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count);
|
||||
// Plus the headset settings panel's quad-layer image when panel is not null, as
|
||||
// aurora_d3d12_set_stereo_targets_with_panel.
|
||||
bool aurora_vulkan_win32_set_targets_with_panel(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count,
|
||||
const AuroraD3D12StereoTarget* panel);
|
||||
bool aurora_vulkan_win32_cancel(uint64_t token);
|
||||
bool aurora_vulkan_win32_disable();
|
||||
void* aurora_vulkan_win32_lock_queue();
|
||||
void aurora_vulkan_win32_unlock_queue(void* guard);
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
+184
-30
@@ -28,6 +28,7 @@
|
||||
#include "gfx/pipeline_cache.hpp"
|
||||
#endif
|
||||
#if defined(__ANDROID__)
|
||||
#include <pthread.h>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include "system_info.hpp"
|
||||
@@ -69,6 +70,9 @@ AuroraConfig g_config;
|
||||
uint32_t g_sdlCustomEventsStart;
|
||||
char g_gameName[4];
|
||||
std::atomic<AuroraFrameWorkerWaitCallback> g_frameWorkerWaitCallback{nullptr};
|
||||
// aurora_set_host_event_pump(): the host pumps SDL itself, from the window's thread.
|
||||
std::atomic_bool g_hostEventPump{false};
|
||||
std::atomic<AuroraFrameLogCallback> g_frameLogCallback{nullptr};
|
||||
// Presentation schedule for the frame being sealed, set by the producer. Jobs carry absolute
|
||||
// deadlines derived from it, so the presenter cannot drift. Zero means present when ready.
|
||||
std::atomic<uint64_t> g_presentScheduleBaseNanos{0};
|
||||
@@ -109,6 +113,9 @@ struct StereoSceneAnchor {
|
||||
};
|
||||
bool active = false;
|
||||
uint32_t localPlayerCount = 1;
|
||||
// World units per metre the anchor was built with, or zero when the packet's
|
||||
// own scale applies (aurora_set_stereo_scene_anchor_scaled).
|
||||
float unitsPerMeter = 0.f;
|
||||
};
|
||||
// Producer thread only, between aurora_set_stereo_scene_anchor() and the seal
|
||||
// that consumes it. Cleared at every seal so a producer that stops publishing
|
||||
@@ -266,7 +273,8 @@ enum class ImGuiFramePolicy {
|
||||
bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy = ImGuiFramePolicy::Immediate,
|
||||
bool* imguiNewFrameOwed = nullptr) noexcept;
|
||||
bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewFrameOwed) noexcept;
|
||||
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor) noexcept;
|
||||
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor,
|
||||
imgui::HostFramePtr hostImGuiFrame) noexcept;
|
||||
|
||||
// The two publication points of a frame-worker cycle, cleared together under `mutex`. Sealed:
|
||||
// producer-shared renderer state is free again. Done: slots encoded, presented, ImGui restarted.
|
||||
@@ -288,6 +296,7 @@ struct FrameWorkerState {
|
||||
// belong to that exact queued frame, not to the producer's next frame.
|
||||
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
|
||||
StereoSceneAnchor sceneAnchor{};
|
||||
imgui::HostFramePtr hostImGuiFrame;
|
||||
// Readiness is polled thousands of times per frame, so these flags double as a publication
|
||||
// barrier. `sealed` is released before `ready`, and both are cleared under `mutex`.
|
||||
std::atomic_bool sealed{true};
|
||||
@@ -338,7 +347,7 @@ bool frame_worker_requested() noexcept {
|
||||
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
// Returns false when a stop request was observed mid-cycle.
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
|
||||
const StereoSceneAnchor& sceneAnchor) noexcept;
|
||||
void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept;
|
||||
#endif
|
||||
@@ -350,6 +359,9 @@ void frame_worker_main() noexcept {
|
||||
}
|
||||
#if defined(__ANDROID__)
|
||||
g_frameWorkerNativeThreadId.store(static_cast<uint32_t>(gettid()), std::memory_order_release);
|
||||
// A thread inherits its creator's name, and the producer that starts this
|
||||
// worker may itself be a named thread; profiles should tell the two apart.
|
||||
pthread_setname_np(pthread_self(), "aurora worker");
|
||||
#endif
|
||||
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
@@ -361,6 +373,7 @@ void frame_worker_main() noexcept {
|
||||
for (;;) {
|
||||
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
|
||||
StereoSceneAnchor sceneAnchor{};
|
||||
imgui::HostFramePtr hostImGuiFrame;
|
||||
bool stereoOnly = false;
|
||||
{
|
||||
std::unique_lock lock(g_frameWorker.mutex);
|
||||
@@ -377,6 +390,7 @@ void frame_worker_main() noexcept {
|
||||
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
|
||||
sceneAnchor = g_frameWorker.sceneAnchor;
|
||||
g_frameWorker.sceneAnchor = {};
|
||||
hostImGuiFrame = std::move(g_frameWorker.hostImGuiFrame);
|
||||
g_frameWorker.jobPending = false;
|
||||
}
|
||||
|
||||
@@ -396,7 +410,7 @@ void frame_worker_main() noexcept {
|
||||
g_frameWorker.cv.notify_all();
|
||||
continue;
|
||||
}
|
||||
if (!run_frame_worker_cycle(sealedFrame, contentTag, sceneAnchor)) {
|
||||
if (!run_frame_worker_cycle(sealedFrame, contentTag, std::move(hostImGuiFrame), sceneAnchor)) {
|
||||
break;
|
||||
}
|
||||
#else
|
||||
@@ -619,7 +633,7 @@ void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height
|
||||
const uint32_t samples = webgpu::g_graphicsConfig.msaaSamples;
|
||||
if (target.color.texture && target.requestedWidth == width && target.requestedHeight == height &&
|
||||
target.samples == samples && target.colorFormat == webgpu::g_graphicsConfig.surfaceConfiguration.format &&
|
||||
target.depthFormat == webgpu::g_graphicsConfig.depthFormat) {
|
||||
target.depthFormat == wgpu::TextureFormat::Depth24PlusStencil8) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -629,19 +643,21 @@ void ensure_stereo_eye_target(uint32_t eyeIndex, uint32_t width, uint32_t height
|
||||
target.resolvedColor = webgpu::create_render_texture(target.color.size.width, target.color.size.height, false);
|
||||
}
|
||||
|
||||
// The cockpit uses one stencil bit to survive later depth-disabled HUD draws.
|
||||
// Keep this attachment eye-only; native EFB depth sampling is unchanged.
|
||||
const wgpu::TextureDescriptor depthDescriptor{
|
||||
.label = eyeIndex == 0 ? "Stereo left eye depth" : "Stereo right eye depth",
|
||||
.usage = wgpu::TextureUsage::RenderAttachment,
|
||||
.dimension = wgpu::TextureDimension::e2D,
|
||||
.size = target.color.size,
|
||||
.format = webgpu::g_graphicsConfig.depthFormat,
|
||||
.format = wgpu::TextureFormat::Depth24PlusStencil8,
|
||||
.mipLevelCount = 1,
|
||||
.sampleCount = samples,
|
||||
};
|
||||
target.depth.texture = g_device.CreateTexture(&depthDescriptor);
|
||||
target.depth.view = target.depth.texture.CreateView();
|
||||
target.depth.size = target.color.size;
|
||||
target.depth.format = webgpu::g_graphicsConfig.depthFormat;
|
||||
target.depth.format = wgpu::TextureFormat::Depth24PlusStencil8;
|
||||
target.requestedWidth = width;
|
||||
target.requestedHeight = height;
|
||||
target.samples = samples;
|
||||
@@ -702,6 +718,27 @@ std::optional<AuroraStereoFrame> request_stereo_frame(uint32_t logicalFrame, uin
|
||||
return std::nullopt;
|
||||
}
|
||||
}
|
||||
// The cockpit overlay is optional: a bad one is dropped, never the frame.
|
||||
if (frame.cockpit.active) {
|
||||
const auto& cockpit = frame.cockpit;
|
||||
bool valid = finite(&cockpit.wheelAngle, 1) && finite(&cockpit.handlebarRadius, 1) &&
|
||||
finite(&cockpit.unitsPerMeter, 1) && cockpit.unitsPerMeter > 0.f &&
|
||||
finite(cockpit.seatFromHandlebar, 12);
|
||||
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
|
||||
valid = valid && finite(cockpit.eyeFromSeat[eye], 12);
|
||||
}
|
||||
for (const auto& hand : cockpit.hands) {
|
||||
valid = valid && finite(&hand.squeeze, 1) && finite(hand.seatFromGrip, 12);
|
||||
}
|
||||
if (!valid) {
|
||||
static bool cockpitRejectionLogged = false;
|
||||
if (!cockpitRejectionLogged) {
|
||||
cockpitRejectionLogged = true;
|
||||
Log.warn("Stereo frame {} carries a non-finite VR cockpit; drawing it without the cockpit", logicalFrame);
|
||||
}
|
||||
frame.cockpit = {};
|
||||
}
|
||||
}
|
||||
return frame;
|
||||
}
|
||||
|
||||
@@ -709,6 +746,16 @@ gfx::StereoReplayFrame make_stereo_replay_frame(const AuroraStereoFrame& input,
|
||||
Mat3x4<float> anchorFromScene;
|
||||
std::memcpy(&anchorFromScene, sceneAnchor.anchorFromScene.data(), sizeof(anchorFromScene));
|
||||
gfx::StereoReplayFrame replay{};
|
||||
replay.cockpit = input.cockpit;
|
||||
// The sealed guest frame owns its scale. The packet may have been sampled
|
||||
// just before a change of scale (a character swap, a lightning strike), so
|
||||
// only its head/IPD translation is rescaled to the frame's.
|
||||
const float frameUnits = sceneAnchor.active && sceneAnchor.unitsPerMeter > 0.f ? sceneAnchor.unitsPerMeter
|
||||
: input.cockpit.unitsPerMeter;
|
||||
const float unitRatio = input.cockpit.unitsPerMeter > 0.f && frameUnits > 0.f
|
||||
? frameUnits / input.cockpit.unitsPerMeter
|
||||
: 1.f;
|
||||
replay.cockpit.unitsPerMeter = frameUnits;
|
||||
for (uint32_t eye = 0; eye < AURORA_STEREO_EYE_COUNT; ++eye) {
|
||||
ensure_stereo_eye_target(eye, input.eyes[eye].width, input.eyes[eye].height);
|
||||
const auto& owned = g_stereoEyeTargets[eye];
|
||||
@@ -723,9 +770,15 @@ gfx::StereoReplayFrame make_stereo_replay_frame(const AuroraStereoFrame& input,
|
||||
.copySourceDepthView = owned.depth.view,
|
||||
.size = owned.color.size,
|
||||
.msaaSamples = webgpu::g_graphicsConfig.msaaSamples,
|
||||
.depthFormat = owned.depth.format,
|
||||
};
|
||||
std::memcpy(&view.projection, input.eyes[eye].projection, sizeof(view.projection));
|
||||
std::memcpy(&view.viewFromCenter, input.eyes[eye].viewFromCenter, sizeof(view.viewFromCenter));
|
||||
if (unitRatio != 1.f) {
|
||||
view.viewFromCenter.m0[3] *= unitRatio;
|
||||
view.viewFromCenter.m1[3] *= unitRatio;
|
||||
view.viewFromCenter.m2[3] *= unitRatio;
|
||||
}
|
||||
// World draws already carry the recorded camera, so they need the anchor
|
||||
// folded in; the virtual screen is authored in the anchored camera's space
|
||||
// and keeps viewFromCenter.
|
||||
@@ -756,6 +809,7 @@ void encode_virtual_screen_eye(wgpu::CommandEncoder& encoder, const webgpu::Pres
|
||||
.label = eyeIndex == 0 ? "Virtual screen left eye" : "Virtual screen right eye",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::VirtualScreen),
|
||||
};
|
||||
{
|
||||
const auto pass = encoder.BeginRenderPass(&descriptor);
|
||||
@@ -1274,6 +1328,7 @@ bool present_presentation_job(const PresentationJob& job) {
|
||||
.label = "Presentation copy pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Present),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetPipeline(webgpu::g_CopyPipeline);
|
||||
@@ -1554,7 +1609,7 @@ void publish_stereo_screen_aspects(const webgpu::PresentSource& presentSource, c
|
||||
// gfx::begin_frame() may already have cleared the display-copy override.
|
||||
void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const webgpu::PresentSource& presentSource,
|
||||
const PresentationImage& image, bool includeImGui,
|
||||
MirrorPlan plan = MirrorPlan::Mono) {
|
||||
MirrorPlan plan = MirrorPlan::Mono, const ImDrawData* hostImGuiData = nullptr) {
|
||||
ZoneScoped;
|
||||
auto viewport = webgpu::calculate_present_viewport(image.texture.size.width, image.texture.size.height,
|
||||
presentSource.size.width, presentSource.size.height);
|
||||
@@ -1576,6 +1631,7 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
|
||||
.label = "Interpolation snapshot pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
const auto imageWidth = static_cast<float>(image.texture.size.width);
|
||||
@@ -1627,11 +1683,16 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
|
||||
.label = "Snapshot ImGui pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Snapshot),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
|
||||
static_cast<float>(image.texture.size.height), 0.f, 1.f);
|
||||
imgui::render(pass);
|
||||
if (hostImGuiData != nullptr) {
|
||||
imgui::render(pass, hostImGuiData);
|
||||
} else {
|
||||
imgui::render(pass);
|
||||
}
|
||||
pass.End();
|
||||
}
|
||||
}
|
||||
@@ -1674,7 +1735,7 @@ bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy, bool* imgui
|
||||
ZoneScoped;
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
webgpu::fail_if_device_lost();
|
||||
if (pumpEvents) {
|
||||
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
|
||||
window::pump_events();
|
||||
}
|
||||
const bool surfaceReconfigurePending = g_surfaceReconfigurePending.load(std::memory_order_acquire);
|
||||
@@ -1731,7 +1792,9 @@ bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewF
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
// Note the debt before gfx::begin_frame() can fail: the synchronous path always started the
|
||||
// ImGui frame here, and the runtime's retry loop depends on that pairing.
|
||||
if (imguiPolicy == ImGuiFramePolicy::Immediate) {
|
||||
if (imgui::host_frames_active()) {
|
||||
// The host starts its own ImGui frames (imgui::host_frame_begin).
|
||||
} else if (imguiPolicy == ImGuiFramePolicy::Immediate) {
|
||||
imgui::new_frame(window::get_window_size());
|
||||
} else if (imguiNewFrameOwed != nullptr) {
|
||||
*imguiNewFrameOwed = true;
|
||||
@@ -1767,8 +1830,13 @@ struct SealedFrameContext {
|
||||
std::optional<AuroraStereoFrame> stereoInput;
|
||||
bool retainStereo = false;
|
||||
imgui::StereoOverlay stereoOverlay;
|
||||
imgui::HostFramePtr imguiFrame;
|
||||
};
|
||||
|
||||
const ImDrawData* host_imgui_data(const SealedFrameContext& ctx) noexcept {
|
||||
return ctx.imguiFrame ? imgui::host_frame_draw_data(*ctx.imguiFrame) : nullptr;
|
||||
}
|
||||
|
||||
// Worker-owned scene state. A separate buffer generation check protects against
|
||||
// synchronous EFB submissions overwriting the retained frame's GPU data.
|
||||
struct RetainedStereoContext {
|
||||
@@ -1827,8 +1895,11 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
|
||||
// Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and
|
||||
// a FIFO already drained into the recorded pass list.
|
||||
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
|
||||
const StereoSceneAnchor& sceneAnchor) {
|
||||
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) {
|
||||
ZoneScopedN("Seal frame");
|
||||
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
|
||||
// one frame; encode_sealed_frame resolves it on its last submission.
|
||||
gfx::gpu_timing_begin_frame();
|
||||
const auto encoderDescriptor = wgpu::CommandEncoderDescriptor{
|
||||
.label = "Redraw encoder",
|
||||
};
|
||||
@@ -1882,7 +1953,12 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u
|
||||
ctx.presentSource = webgpu::current_present_source();
|
||||
// ImGui draw lists are built once per frame and replayed by each slot's ImGui pass, which is why
|
||||
// the next ImGui frame cannot start until the encode phase is done.
|
||||
imgui::render_frame_data();
|
||||
if (hostImGuiFrame) {
|
||||
// The host closed its own ImGui frame and handed over a copy of the draw data.
|
||||
ctx.imguiFrame = std::move(hostImGuiFrame);
|
||||
} else {
|
||||
imgui::render_frame_data();
|
||||
}
|
||||
// The headset panel's draw data follows the same rule on the host's side.
|
||||
ctx.stereoOverlay = imgui::latch_stereo_overlay();
|
||||
// Drop the sealed frame's lazy RAM-readback requests while the producer is still excluded; it
|
||||
@@ -1974,7 +2050,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
|
||||
gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false);
|
||||
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
|
||||
presentationJobs.push_back({
|
||||
.image = std::move(image),
|
||||
.logicalFrame = ctx.logicalFrame,
|
||||
@@ -1988,14 +2064,25 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
|
||||
// A demanded CPU-visible EFB readback submits a prefix of the frame, so replaying the resumed
|
||||
// stream would mutate an already-rendered EFB. Render once, then duplicate into the slots.
|
||||
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo);
|
||||
//
|
||||
// On a headset an immersive frame's native render is never presented: the eyes replay the draws
|
||||
// themselves and only sample the EFB copies it resolves. So it stops after the last pass that
|
||||
// produces one of those copies (never the display copy), which on a Quest 3 was 4 to 6 ms of a
|
||||
// 12 ms GPU frame spent on a 1280x720 image nobody saw. A pending CPU readback or a frame
|
||||
// capture still gets the whole image.
|
||||
int32_t nativeRenderLastPass = INT32_MAX;
|
||||
if (headsetOnly && immersiveReplay && !gfx::efb_ram::has_pending() &&
|
||||
g_captureFrame.load(std::memory_order_acquire) == UINT32_MAX) {
|
||||
nativeRenderLastPass = gfx::last_pass_feeding_replay(sealedFrame);
|
||||
}
|
||||
gfx::render(sealedFrame, encoder, -1, !immersiveReplay && !ctx.retainStereo, nativeRenderLastPass);
|
||||
// The copy targets now hold this frame's resolves, so queue their readbacks on the same encoder;
|
||||
// completion is harvested in gfx::after_submit, never waited on here.
|
||||
gfx::efb_ram::encode_async_downloads(encoder);
|
||||
if (!ctx.replayInterpolatedFrames) {
|
||||
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
|
||||
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
|
||||
presentationJobs.push_back({
|
||||
.image = std::move(image),
|
||||
.logicalFrame = ctx.logicalFrame,
|
||||
@@ -2035,7 +2122,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
// showing. Black re-clears it below, once the eyes have taken their copy.
|
||||
const bool virtualScreenNeedsMono = stereoOutput && !immersiveReplay;
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true,
|
||||
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan);
|
||||
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan, host_imgui_data(ctx));
|
||||
if (stereoOutput) {
|
||||
publish_stereo_screen_aspects(ctx.presentSource, finalImage->texture.size, immersiveReplay);
|
||||
}
|
||||
@@ -2057,7 +2144,8 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
stereo_overlay::composite_flat(encoder, output.view, output.size, eye);
|
||||
}
|
||||
if (mirrorPlan == MirrorPlan::Black && !headsetOnly) {
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black,
|
||||
host_imgui_data(ctx));
|
||||
}
|
||||
}
|
||||
if (stereoOutput) {
|
||||
@@ -2070,7 +2158,9 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
.presentAt = slotPresentDeadline(ctx.interpolatedFrameCount),
|
||||
.interpolated = false,
|
||||
});
|
||||
gfx::gpu_timing_end_frame(encoder);
|
||||
submitEncodedSlot(encoder, pendingStereoSink ? &*pendingStereoSink : nullptr);
|
||||
gfx::gpu_timing_after_submit();
|
||||
|
||||
// A group that finished encoding past its anchor slides forward by whole display periods, never
|
||||
// per slot. The cursor keeps two groups off one anchor, which bursts then holds for a period.
|
||||
@@ -2180,11 +2270,22 @@ void record_frame_telemetry() {
|
||||
{
|
||||
// `adb shell setprop debug.wiicompiled.fpslog 1` before launch logs the game's rendered frame rate every five
|
||||
// seconds. The headset compositor's own log (logcat tag VrApi) repeats frames, so it cannot show this.
|
||||
static const bool fpsLog = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
|
||||
static const bool fpsLog = [] {
|
||||
const bool on = android_debug::property_int("debug.wiicompiled.fpslog", 0) == 1;
|
||||
// The same switch turns on the per-pass GPU timestamps reported below the frame-rate line.
|
||||
gfx::gpu_timing_set_enabled(on);
|
||||
return on;
|
||||
}();
|
||||
if (fpsLog) {
|
||||
static auto windowStart = std::chrono::steady_clock::now();
|
||||
static uint32_t windowFrames = 0;
|
||||
// Draw calls are what a recorded frame costs three times over (mono and both eyes), so they belong beside the
|
||||
// GPU timings: an overlay that stops draws merging shows up here long before it shows up as a frame rate.
|
||||
static uint64_t windowDraws = 0;
|
||||
static uint64_t windowMerged = 0;
|
||||
++windowFrames;
|
||||
windowDraws += gfx::g_stats.drawCallCount;
|
||||
windowMerged += gfx::g_stats.mergedDrawCallCount;
|
||||
const auto now = std::chrono::steady_clock::now();
|
||||
const std::chrono::duration<double> elapsed = now - windowStart;
|
||||
if (elapsed.count() >= 5.0) {
|
||||
@@ -2200,9 +2301,25 @@ void record_frame_telemetry() {
|
||||
const double encode = msPerFrame(g_workerEncodeNs);
|
||||
Log.info("Game frame rate {:.1f} FPS ({} frames in {:.2f} s); per frame the producer waited {:.2f} ms for "
|
||||
"DONE and {:.2f} ms for SEALED; the worker spent {:.2f} ms sealing, {:.2f} ms waiting for the "
|
||||
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding",
|
||||
"prepare permit, {:.2f} ms preparing the next frame and {:.2f} ms encoding; {:.0f} draw calls a "
|
||||
"frame ({:.0f} primitives merged away)",
|
||||
windowFrames / elapsed.count(), windowFrames, elapsed.count(), waitDone, waitSealed, seal,
|
||||
permitWait, prepare, encode);
|
||||
permitWait, prepare, encode,
|
||||
static_cast<double>(windowDraws) / std::max(windowFrames, 1u),
|
||||
static_cast<double>(windowMerged) / std::max(windowFrames, 1u));
|
||||
windowDraws = 0;
|
||||
windowMerged = 0;
|
||||
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
|
||||
Log.info("{}", gpuTiming);
|
||||
}
|
||||
if (const auto frameLog = g_frameLogCallback.load(std::memory_order_acquire)) {
|
||||
char extra[512];
|
||||
extra[0] = '\0';
|
||||
frameLog(extra, sizeof(extra), elapsed.count(), windowFrames);
|
||||
if (extra[0] != '\0') {
|
||||
Log.info("{}", extra);
|
||||
}
|
||||
}
|
||||
windowStart = now;
|
||||
windowFrames = 0;
|
||||
}
|
||||
@@ -2214,7 +2331,7 @@ void record_frame_telemetry() {
|
||||
|
||||
// One complete frame-worker cycle. Desktop and headset interpolation both
|
||||
// release the producer after sealing, before encoding their extra scene views.
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
|
||||
const StereoSceneAnchor& sceneAnchor) noexcept {
|
||||
ZoneScopedN("Frame worker cycle");
|
||||
webgpu::fail_if_device_lost();
|
||||
@@ -2226,7 +2343,7 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
auto stretchStarted = std::chrono::steady_clock::now();
|
||||
{
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
|
||||
}
|
||||
g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
|
||||
stretchStarted = std::chrono::steady_clock::now();
|
||||
@@ -2283,11 +2400,11 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
// Synchronous frame submission: seal, encode and present inline on the calling thread. Used when
|
||||
// the frame worker is disabled (RenderDoc captures) and on the boot path.
|
||||
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
|
||||
const StereoSceneAnchor& sceneAnchor) noexcept {
|
||||
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) noexcept {
|
||||
ZoneScoped;
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
webgpu::fail_if_device_lost();
|
||||
if (pumpEvents) {
|
||||
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
|
||||
window::pump_events();
|
||||
}
|
||||
gfx::SealedFrame sealedFrame;
|
||||
@@ -2298,7 +2415,7 @@ void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
|
||||
if (drainFifo) {
|
||||
gx::fifo::drain();
|
||||
}
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
|
||||
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
|
||||
}
|
||||
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
|
||||
@@ -2324,7 +2441,9 @@ bool begin_frame() noexcept {
|
||||
ensure_frame_worker_started();
|
||||
// SDL needs event pumping on the window-owning producer thread, and the worker passes
|
||||
// pumpEvents=false, so keep it here even when the fast path returns early.
|
||||
window::pump_events();
|
||||
if (!g_hostEventPump.load(std::memory_order_acquire)) {
|
||||
window::pump_events();
|
||||
}
|
||||
bool waitForSurfacePreparation = false;
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
// A surface mutation can legitimately fail preparation, and optimistic success would let GX/ImGui
|
||||
@@ -2374,7 +2493,7 @@ bool begin_frame() noexcept {
|
||||
return prepared;
|
||||
}
|
||||
|
||||
void end_frame(uint64_t contentTag) noexcept {
|
||||
void end_frame(uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame) noexcept {
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
webgpu::fail_if_device_lost();
|
||||
#endif
|
||||
@@ -2385,7 +2504,7 @@ void end_frame(uint64_t contentTag) noexcept {
|
||||
g_pendingStereoLocalPlayerCount = 1;
|
||||
g_pendingSceneAnchor = {};
|
||||
if (!frame_worker_requested()) {
|
||||
end_frame_impl(true, true, contentTag, sceneAnchor);
|
||||
end_frame_impl(true, true, contentTag, sceneAnchor, std::move(hostImGuiFrame));
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2407,6 +2526,7 @@ void end_frame(uint64_t contentTag) noexcept {
|
||||
g_frameWorker.ready.store(false, std::memory_order_release);
|
||||
g_frameWorker.contentTag = contentTag;
|
||||
g_frameWorker.sceneAnchor = sceneAnchor;
|
||||
g_frameWorker.hostImGuiFrame = std::move(hostImGuiFrame);
|
||||
g_frameWorker.jobPending = true;
|
||||
g_frameWorker.prepareAllowed = false;
|
||||
}
|
||||
@@ -2509,11 +2629,45 @@ AuroraInfo aurora_initialize(int argc, char* argv[], const AuroraConfig* config)
|
||||
void aurora_shutdown() { aurora::shutdown(); }
|
||||
const AuroraEvent* aurora_update() { return aurora::update(); }
|
||||
bool aurora_begin_frame() { return aurora::begin_frame(); }
|
||||
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN); }
|
||||
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag); }
|
||||
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN, {}); }
|
||||
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag, {}); }
|
||||
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame) {
|
||||
aurora::imgui::HostFramePtr frame;
|
||||
if (imguiFrame != nullptr) {
|
||||
auto* holder = static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
|
||||
frame = std::move(*holder);
|
||||
delete holder;
|
||||
}
|
||||
aurora::end_frame(contentTag, std::move(frame));
|
||||
}
|
||||
void aurora_set_host_event_pump(bool hostPumps) {
|
||||
aurora::g_hostEventPump.store(hostPumps, std::memory_order_release);
|
||||
}
|
||||
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback) {
|
||||
aurora::g_frameLogCallback.store(callback, std::memory_order_release);
|
||||
}
|
||||
extern "C" void aurora_imgui_host_frame_begin(void) {
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
// ImGui's WebGPU backend creates its device objects lazily from new_frame.
|
||||
std::lock_guard gpuLock(aurora::g_rendererGpuMutex);
|
||||
#endif
|
||||
aurora::imgui::host_frame_begin(aurora::window::get_window_size());
|
||||
}
|
||||
extern "C" void* aurora_imgui_host_frame_end(void) { return new aurora::imgui::HostFramePtr(aurora::imgui::host_frame_end()); }
|
||||
extern "C" void aurora_imgui_host_frame_release(void* imguiFrame) {
|
||||
delete static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
|
||||
}
|
||||
void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]) {
|
||||
aurora::set_stereo_scene_anchor(anchorFromScene);
|
||||
}
|
||||
void aurora_set_stereo_scene_anchor_scaled(const float anchorFromScene[12], float unitsPerMeter) {
|
||||
aurora::set_stereo_scene_anchor(anchorFromScene);
|
||||
uint32_t bits = 0;
|
||||
std::memcpy(&bits, &unitsPerMeter, sizeof(bits));
|
||||
if (aurora::g_pendingSceneAnchor.active && (bits & 0x7f800000u) != 0x7f800000u && unitsPerMeter > 0.f) {
|
||||
aurora::g_pendingSceneAnchor.unitsPerMeter = unitsPerMeter;
|
||||
}
|
||||
}
|
||||
void aurora_set_stereo_local_player_count(uint32_t count) {
|
||||
aurora::g_pendingStereoLocalPlayerCount = count >= 1 && count <= 4 ? count : 1;
|
||||
}
|
||||
|
||||
@@ -10,10 +10,68 @@
|
||||
|
||||
#include "../../gfx/common.hpp"
|
||||
#include "../../gx/fifo.hpp"
|
||||
#include "../../gx/native_wheel.hpp"
|
||||
|
||||
// GX-thread entry points for the VR native steering wheel (native_wheel.hpp).
|
||||
// The runtime posts these in order with the frame's draws.
|
||||
extern "C" void aurora_clear_native_wheel_vertices() {
|
||||
// Clearing an empty set reports nothing, so a host that clears both before and after a frame's draws keeps
|
||||
// that frame's count.
|
||||
if (aurora::gx::nativeWheelArrays.empty()) return;
|
||||
aurora::gx::fifo::drain();
|
||||
aurora::gx::nativeWheelLastMatches.store(aurora::gx::nativeWheelMatches);
|
||||
aurora::gx::nativeWheelMatches = 0;
|
||||
aurora::gx::nativeWheelPreviousSources.clear();
|
||||
for (const auto& array : aurora::gx::nativeWheelArrays)
|
||||
aurora::gx::nativeWheelPreviousSources.push_back(array.source);
|
||||
aurora::gx::native_wheel_report();
|
||||
aurora::gx::nativeWheelArrays.clear();
|
||||
aurora::gx::nativeWheelLastDecision = nullptr;
|
||||
aurora::gx::nativeWheelLastDrawCommand = nullptr;
|
||||
}
|
||||
extern "C" uint32_t aurora_native_wheel_draw_count() { return aurora::gx::nativeWheelLastMatches.load(); }
|
||||
extern "C" void aurora_set_native_wheel_vertices(const void* source, const void* replacement, uint32_t size,
|
||||
const float* modelView) {
|
||||
if (!source || !replacement || !modelView || !size || size > 65536) return;
|
||||
aurora::gx::NativeWheelArray array;
|
||||
array.source = source;
|
||||
const auto* bytes = static_cast<const uint8_t*>(replacement);
|
||||
array.bytes.assign(bytes, bytes + size);
|
||||
std::memcpy(array.modelView.data(), modelView, sizeof(float) * 12);
|
||||
aurora::gx::nativeWheelArrays.push_back(std::move(array));
|
||||
// The vector may have moved its elements, and a set changes what a draw resolves to in any case, so the decision
|
||||
// the merge test compares against is dropped rather than left pointing into the old storage.
|
||||
aurora::gx::nativeWheelLastDecision = nullptr;
|
||||
aurora::gx::nativeWheelLastDrawCommand = nullptr;
|
||||
}
|
||||
|
||||
// Single definition for the `Log` that gx.hpp declares for this directory.
|
||||
aurora::Module Log("aurora::gx");
|
||||
|
||||
namespace aurora::gx {
|
||||
// Called as a set is cleared: a line at about half a second, five seconds and
|
||||
// a minute of sets, with the matrices on the first report that saw a draw.
|
||||
void native_wheel_report() {
|
||||
auto& diagnostics=nativeWheelDiagnostics;
|
||||
++diagnostics.sets;
|
||||
++nativeWheelClears;
|
||||
if(nativeWheelClears!=30 && nativeWheelClears!=300 && nativeWheelClears!=3600) return;
|
||||
::Log.info("Native steering wheel: {} sets; draws binding a replaced array {} ({} with a larger range, {} outside a "
|
||||
"set); matched {}; closest position matrix off by {} ({} matrices)",
|
||||
diagnostics.sets,diagnostics.boundDraws,diagnostics.oversizeDraws,diagnostics.outsideDraws,
|
||||
diagnostics.matchedDraws,diagnostics.bestError,diagnostics.bestIndexed?"indexed":"current");
|
||||
if(diagnostics.boundDraws!=0 && nativeWheelReports++==0) {
|
||||
const auto& m=diagnostics.bestMatrix;
|
||||
const auto& e=diagnostics.expected;
|
||||
::Log.info("Native steering wheel: closest [{} {} {} {} | {} {} {} {} | {} {} {} {}] expected [{} {} {} {} | {} {} "
|
||||
"{} {} | {} {} {} {}]",
|
||||
m[0],m[1],m[2],m[3],m[4],m[5],m[6],m[7],m[8],m[9],m[10],m[11],
|
||||
e[0],e[1],e[2],e[3],e[4],e[5],e[6],e[7],e[8],e[9],e[10],e[11]);
|
||||
}
|
||||
diagnostics={};
|
||||
}
|
||||
} // namespace aurora::gx
|
||||
|
||||
static void GXWriteString(const char* label) {
|
||||
auto length = strlen(label);
|
||||
|
||||
|
||||
@@ -549,6 +549,13 @@ void GXCopyTex(void* dest, GXBool clear) {
|
||||
.clearAlpha = true,
|
||||
.clearDepth = false,
|
||||
}),
|
||||
.stereoPipeline = aurora::stereo_frame_provider_active() ? aurora::gfx::pipeline_ref(aurora::gfx::clear::PipelineConfig{
|
||||
.msaaSamples = aurora::gfx::get_sample_count(),
|
||||
.clearColor = false,
|
||||
.clearAlpha = true,
|
||||
.clearDepth = false,
|
||||
.stereoStencil = true,
|
||||
}) : 0,
|
||||
.color = wgpu::Color{0.f, 0.f, 0.f, g_gxState.dstAlpha / 255.f},
|
||||
});
|
||||
}
|
||||
|
||||
@@ -105,7 +105,7 @@ fn fs_main() -> FragmentOutput {
|
||||
.targets = &colorTarget,
|
||||
};
|
||||
const wgpu::DepthStencilState depthStencil{
|
||||
.format = g_graphicsConfig.depthFormat,
|
||||
.format = config.stereoStencil ? wgpu::TextureFormat::Depth24PlusStencil8 : g_graphicsConfig.depthFormat,
|
||||
.depthWriteEnabled = config.clearDepth,
|
||||
.depthCompare = wgpu::CompareFunction::Always,
|
||||
};
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
namespace aurora::gfx::clear {
|
||||
struct DrawData {
|
||||
PipelineRef pipeline;
|
||||
PipelineRef stereoPipeline = 0;
|
||||
Range uniformRange;
|
||||
wgpu::Color color;
|
||||
float depth = 0.f;
|
||||
@@ -18,14 +19,14 @@ struct DrawData {
|
||||
ClipRect scissor{};
|
||||
};
|
||||
|
||||
constexpr uint32_t ClearPipelineConfigVersion = 3;
|
||||
constexpr uint32_t ClearPipelineConfigVersion = 4;
|
||||
struct PipelineConfig {
|
||||
uint32_t version = ClearPipelineConfigVersion;
|
||||
uint32_t msaaSamples = 1;
|
||||
bool clearColor = true;
|
||||
bool clearAlpha = true;
|
||||
bool clearDepth = true;
|
||||
uint8_t _pad = 0;
|
||||
bool stereoStencil = false;
|
||||
};
|
||||
static_assert(std::has_unique_object_representations_v<PipelineConfig>);
|
||||
|
||||
|
||||
@@ -0,0 +1,299 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
|
||||
//
|
||||
// VR cockpit overlay: the synthetic steering wheel or handlebar (used when the
|
||||
// vehicle's own wheel cannot be animated) and the tracked hands, drawn per eye
|
||||
// in metres against the replayed scene's depth. See OPENXR.md, "Steering wheel
|
||||
// and hand steering".
|
||||
#pragma once
|
||||
#include "common.hpp"
|
||||
#include "../webgpu/gpu.hpp"
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <vector>
|
||||
|
||||
namespace aurora::gfx::cockpit {
|
||||
using V = std::array<float, 3>;
|
||||
using M = std::array<float, 12>;
|
||||
inline V add(V a, V b) { return {a[0]+b[0], a[1]+b[1], a[2]+b[2]}; }
|
||||
inline V sub(V a, V b) { return {a[0]-b[0], a[1]-b[1], a[2]-b[2]}; }
|
||||
inline V mul(V a, float b) { return {a[0]*b, a[1]*b, a[2]*b}; }
|
||||
inline float dot(V a, V b) { return a[0]*b[0]+a[1]*b[1]+a[2]*b[2]; }
|
||||
inline V cross(V a, V b) { return {a[1]*b[2]-a[2]*b[1],a[2]*b[0]-a[0]*b[2],a[0]*b[1]-a[1]*b[0]}; }
|
||||
inline V norm(V a) { return mul(a, 1/std::sqrt(std::max(dot(a,a), 1e-10f))); }
|
||||
inline V point(const float* m, V p) {
|
||||
return {m[0]*p[0]+m[1]*p[1]+m[2]*p[2]+m[3], m[4]*p[0]+m[5]*p[1]+m[6]*p[2]+m[7],
|
||||
m[8]*p[0]+m[9]*p[1]+m[10]*p[2]+m[11]};
|
||||
}
|
||||
inline M identity() { return {1,0,0,0,0,1,0,0,0,0,1,0}; }
|
||||
inline M compose(const M& a, const M& b) {
|
||||
M result{};
|
||||
for(int r=0;r<3;++r) {
|
||||
for(int c=0;c<3;++c) for(int k=0;k<3;++k) result[r*4+c]+=a[r*4+k]*b[k*4+c];
|
||||
result[r*4+3]=a[r*4+3];
|
||||
for(int k=0;k<3;++k) result[r*4+3]+=a[r*4+k]*b[k*4+3];
|
||||
}
|
||||
return result;
|
||||
}
|
||||
inline M inverse(const M& m) {
|
||||
M out=identity();
|
||||
for(int r=0;r<3;++r) for(int c=0;c<3;++c) out[r*4+c]=m[c*4+r];
|
||||
const auto p=point(out.data(), {-m[3],-m[7],-m[11]});
|
||||
out[3]=p[0];out[7]=p[1];out[11]=p[2];return out;
|
||||
}
|
||||
inline M from_pose(const float* p) {
|
||||
const float x=p[0],y=p[1],z=p[2],w=p[3];
|
||||
return {1-2*(y*y+z*z),2*(x*y-z*w),2*(x*z+y*w),p[4],
|
||||
2*(x*y+z*w),1-2*(x*x+z*z),2*(y*z-x*w),p[5],
|
||||
2*(x*z-y*w),2*(y*z+x*w),1-2*(x*x+y*y),p[6]};
|
||||
}
|
||||
struct HandMesh {
|
||||
std::vector<AuroraVRHandVertex> vertices;
|
||||
std::vector<uint16_t> indices;
|
||||
std::array<M,26> bind{}, inverseBind{};
|
||||
std::array<int32_t,26> parents{};
|
||||
};
|
||||
inline std::mutex meshMutex;
|
||||
inline std::array<std::shared_ptr<const HandMesh>,2> meshes;
|
||||
struct Vertex { V position, color; };
|
||||
inline void triangle(std::vector<Vertex>& vertices, V a, V b, V c, V color) {
|
||||
const V normal=norm(cross(sub(b,a),sub(c,a)));
|
||||
const float light=0.55f+0.45f*std::abs(dot(normal,norm({0.3f,0.8f,0.5f})));
|
||||
color=mul(color,light);
|
||||
vertices.insert(vertices.end(),{{a,color},{b,color},{c,color}});
|
||||
}
|
||||
inline void tube(std::vector<Vertex>& v, V a, V b, float radius, V color, int sides=8) {
|
||||
const auto direction=norm(sub(b,a));
|
||||
const auto u=norm(cross(direction,std::abs(direction[1])<0.9f?V{0,1,0}:V{1,0,0}));
|
||||
const auto w=cross(direction,u);
|
||||
for(int i=0;i<sides;++i) {
|
||||
const float t=float(i)*6.2831853f/sides, t1=float(i+1)*6.2831853f/sides;
|
||||
const V o=mul(add(mul(u,std::cos(t)),mul(w,std::sin(t))),radius);
|
||||
const V p=mul(add(mul(u,std::cos(t1)),mul(w,std::sin(t1))),radius);
|
||||
triangle(v,add(a,o),add(b,o),add(b,p),color);
|
||||
triangle(v,add(a,o),add(b,p),add(a,p),color);
|
||||
triangle(v,a,add(a,p),add(a,o),color);
|
||||
triangle(v,b,add(b,o),add(b,p),color);
|
||||
}
|
||||
}
|
||||
inline void ellipsoid(std::vector<Vertex>& vertices,V center,V radii,V color) {
|
||||
const auto surface=[&](int ring,int segment) {
|
||||
const float latitude=float(ring)*3.14159265f/6,longitude=float(segment)*6.2831853f/12;
|
||||
return add(center,{radii[0]*std::sin(latitude)*std::cos(longitude),radii[1]*std::cos(latitude),
|
||||
radii[2]*std::sin(latitude)*std::sin(longitude)});
|
||||
};
|
||||
for(int ring=0;ring<6;++ring) for(int segment=0;segment<12;++segment) {
|
||||
const auto a=surface(ring,segment),b=surface(ring+1,segment),c=surface(ring+1,segment+1),d=surface(ring,segment+1);
|
||||
if(ring>0) triangle(vertices,a,b,d,color);
|
||||
if(ring<5) triangle(vertices,b,c,d,color);
|
||||
}
|
||||
}
|
||||
// Rounded palm and individually articulated fingers, in the controller's grip
|
||||
// space as OpenXR defines it: the origin is the palm centroid, -Z runs up the
|
||||
// tube the curled fingers form (little finger towards thumb), and +X is normal
|
||||
// to the palm - *away* from it on the left hand, *into* it on the right. That
|
||||
// asymmetry is what makes both grips carry the same orientation when the hands
|
||||
// hold a wheel symmetrically, so the fingers run along -Y on both, and it is
|
||||
// the geometry across the palm that mirrors: fingers close towards +X on the
|
||||
// left hand and -X on the right, with the thumb on the same side. Building the
|
||||
// fingers on any other axis bends them out of the back of the hand (seen on a
|
||||
// Quest 3 on 2026-09-22) or, for the right hand alone, points them at the
|
||||
// player (seen on the PC on 2026-09-23).
|
||||
inline void glove(std::vector<Vertex>& v, const AuroraCockpitHand& hand, int side) {
|
||||
const size_t start=v.size();
|
||||
const V white{0.91f,0.95f,1.0f};
|
||||
const float palm=side==0?1.0f:-1.0f; // hand 0 is the left one
|
||||
const float curl=std::clamp(hand.held?0.85f:hand.squeeze,0.0f,1.0f);
|
||||
// Thin through the palm's normal, a little wider across the knuckles than
|
||||
// the palm is long.
|
||||
ellipsoid(v,{0,0,0},{0.018f,0.043f,0.041f},white);
|
||||
for(int finger=0;finger<4;++finger) {
|
||||
// Index finger nearest the thumb (-Z), little finger last.
|
||||
V a{0.0f,-0.030f,-0.025f+finger*0.017f};
|
||||
const float length=finger==0||finger==3?0.021f:0.026f;
|
||||
for(int joint=0;joint<3;++joint) {
|
||||
const float angle=curl*(0.55f+joint*0.8f);
|
||||
V b=add(a,{palm*std::sin(angle)*length,-std::cos(angle)*length,0.0f});
|
||||
tube(v,a,b,0.008f,white);
|
||||
ellipsoid(v,b,{0.008f,0.008f,0.008f},white);a=b;
|
||||
}
|
||||
}
|
||||
// Thumb: out of the palm's thumb side, closing across the fingers.
|
||||
const V thumbKnuckle{palm*0.026f,-0.034f,-0.030f};
|
||||
tube(v,{palm*0.010f,-0.012f,-0.034f},thumbKnuckle,0.010f,white);
|
||||
tube(v,thumbKnuckle,{palm*(0.030f+0.014f*curl),-(0.052f-0.016f*curl),-0.020f},0.009f,white);
|
||||
for(size_t i=start;i<v.size();++i) v[i].position=point(hand.seatFromGrip,v[i].position);
|
||||
}
|
||||
inline void runtime_hand(std::vector<Vertex>& out, const AuroraCockpitHand& hand, const HandMesh& mesh) {
|
||||
std::array<M,26> posed{}, skin{};
|
||||
std::array<bool,26> done{};
|
||||
const float curl=std::clamp(hand.held?0.85f:hand.squeeze,0.0f,1.0f);
|
||||
// Bind hierarchy is supplied by the runtime. Root and wrist stay rigid;
|
||||
// finger joints curl locally when controllers provide squeeze input.
|
||||
for(int pass=0;pass<26;++pass) for(int j=0;j<26;++j) {
|
||||
if(done[j]) continue;
|
||||
const int parent=mesh.parents[j];
|
||||
if(parent>=0&&parent<26&&!done[parent]) continue;
|
||||
M local=parent>=0&&parent<26?compose(mesh.inverseBind[parent],mesh.bind[j]):mesh.bind[j];
|
||||
const bool fingerJoint=j>=2 && j!=6 && j!=11 && j!=16 && j!=21;
|
||||
if(fingerJoint) {
|
||||
// OpenXR joints point -Z toward the fingertip and +Y out of the back
|
||||
// of the hand. Flexion is therefore negative about local X, for both
|
||||
// hands; positive angles bend the fingers backward on runtime meshes.
|
||||
const float a=-curl*(j<6?0.3f:0.75f),c=std::cos(a),s=std::sin(a);
|
||||
local=compose(local,M{1,0,0,0,0,c,-s,0,0,s,c,0});
|
||||
}
|
||||
posed[j]=parent>=0&&parent<26?compose(posed[parent],local):local;
|
||||
skin[j]=compose(mesh.inverseBind[1],compose(posed[j],mesh.inverseBind[j]));
|
||||
done[j]=true;
|
||||
}
|
||||
std::vector<V> points(mesh.vertices.size());
|
||||
for(size_t i=0;i<points.size();++i) {
|
||||
const auto& v=mesh.vertices[i]; V p{}; float total=0;
|
||||
for(int w=0;w<4;++w) if(v.joints[w]>=0&&v.joints[w]<26&&done[v.joints[w]]&&v.weights[w]>0) {
|
||||
p=add(p,mul(point(skin[v.joints[w]].data(),{v.position[0],v.position[1],v.position[2]}),v.weights[w]));
|
||||
total+=v.weights[w];
|
||||
}
|
||||
if(total>0) p=mul(p,1/total);
|
||||
p=add(p,{0,0,0.04f}); // wrist behind the controller grip/palm origin.
|
||||
points[i]=point(hand.seatFromGrip,p);
|
||||
}
|
||||
for(size_t i=0;i+2<mesh.indices.size();i+=3)
|
||||
triangle(out,points[mesh.indices[i]],points[mesh.indices[i+1]],points[mesh.indices[i+2]],{0.91f,0.95f,1.0f});
|
||||
}
|
||||
inline void build_geometry(const AuroraCockpit& cockpit, std::vector<Vertex>& vertices) {
|
||||
vertices.clear();vertices.reserve(12000);
|
||||
// The visible radius and position must match runtime/vr/steering_wheel.h.
|
||||
if (!cockpit.nativeWheel && cockpit.bike) {
|
||||
const float c=std::cos(cockpit.wheelAngle),s=std::sin(cockpit.wheelAngle);
|
||||
const auto barPoint=[&](float x,float y,float z) {
|
||||
return point(cockpit.seatFromHandlebar,{c*x+s*y,-s*x+c*y,z});
|
||||
};
|
||||
const float radius=cockpit.handlebarRadius;
|
||||
tube(vertices,barPoint(-radius,0,0),barPoint(radius,0,0),0.013f,{0.45f,0.48f,0.52f});
|
||||
for(float side:{-1.0f,1.0f})
|
||||
tube(vertices,barPoint(side*std::max(radius-0.10f,0.0f),0,0),barPoint(side*radius,0,0),0.024f,{0.12f,0.18f,0.19f});
|
||||
tube(vertices,barPoint(0,0,-0.13f),barPoint(0,0,0),0.023f,{0.12f,0.65f,0.61f});
|
||||
} else if (!cockpit.nativeWheel) {
|
||||
const auto rim=[&](float angle) -> V { return {0.18f*std::cos(angle),-0.30f+0.18f*std::sin(angle),-0.42f}; };
|
||||
for(int i=0;i<64;++i) {
|
||||
const float angle=float(i)*6.2831853f/64-cockpit.wheelAngle;
|
||||
const V color=i>=15&&i<=17?V{0.2f,0.9f,0.8f}:V{0.14f,0.17f,0.20f};
|
||||
tube(vertices,rim(angle),rim(angle+6.2831853f/64),0.016f,color,6);
|
||||
}
|
||||
for(float a : {0.0f,3.14159265f,4.71238898f})
|
||||
tube(vertices,{0,-0.30f,-0.42f},rim(a-cockpit.wheelAngle),0.011f,{0.45f,0.48f,0.52f});
|
||||
tube(vertices,{0,-0.30f,-0.445f},{0,-0.30f,-0.395f},0.035f,{0.12f,0.65f,0.61f},16);
|
||||
}
|
||||
std::array<std::shared_ptr<const HandMesh>,2> current;
|
||||
{ std::lock_guard lock(meshMutex);current=meshes; }
|
||||
for(int side=0;side<2;++side) if(cockpit.hands[side].tracked) {
|
||||
if(current[side]) runtime_hand(vertices,cockpit.hands[side],*current[side]);
|
||||
else glove(vertices,cockpit.hands[side],side);
|
||||
}
|
||||
}
|
||||
inline std::vector<Vertex> geometry(const AuroraCockpit& cockpit) {
|
||||
std::vector<Vertex> result;build_geometry(cockpit,result);return result;
|
||||
}
|
||||
inline std::atomic<uint64_t> meshRevision{1};
|
||||
inline std::vector<Vertex> frameVertices;
|
||||
inline AuroraCockpit cachedCockpit{};
|
||||
inline uint64_t cachedMeshRevision=0;
|
||||
inline wgpu::RenderPipeline pipeline;
|
||||
struct SceneDepth {
|
||||
float z=0, constant=0;
|
||||
bool valid=false;
|
||||
};
|
||||
inline uint32_t pipelineSamples=0;
|
||||
inline bool pipelineReversedDepth=false;
|
||||
inline wgpu::TextureFormat pipelineFormat{}, pipelineDepthFormat{};
|
||||
inline std::array<wgpu::Buffer,2> vertexBuffers;
|
||||
inline std::array<uint64_t,2> vertexCapacity{};
|
||||
inline void shutdown() { pipeline=nullptr;pipelineSamples=0;vertexBuffers={};vertexCapacity={};cachedMeshRevision=0;frameVertices.clear(); }
|
||||
inline void render(wgpu::CommandEncoder& cmd,const StereoReplayFrame& frame,uint32_t eye,SceneDepth sceneDepth={},
|
||||
const wgpu::RenderPassEncoder* existingPass=nullptr) {
|
||||
if(!frame.cockpit.active || !sceneDepth.valid) return;
|
||||
using namespace webgpu;
|
||||
const auto& target=frame.eyes[eye].target;
|
||||
const auto format=g_graphicsConfig.surfaceConfiguration.format;
|
||||
// The guest can reverse its viewport depth independently of Aurora's
|
||||
// global reversed-Z convention. The final 1/d coefficient is authoritative.
|
||||
const bool reversedDepth=sceneDepth.constant>0;
|
||||
if(!pipeline||pipelineSamples!=target.msaaSamples||pipelineFormat!=format||pipelineReversedDepth!=reversedDepth||pipelineDepthFormat!=target.depthFormat) {
|
||||
wgpu::ShaderSourceWGSL source{};
|
||||
source.code=R"(
|
||||
struct Out { @builtin(position) position: vec4f, @location(0) color: vec3f };
|
||||
@vertex fn vs(@location(0) position: vec4f, @location(1) color: vec3f) -> Out {
|
||||
var o: Out; o.position=position; o.color=color; return o;
|
||||
}
|
||||
@fragment fn fs(i: Out) -> @location(0) vec4f { return vec4f(i.color,1); }
|
||||
)";
|
||||
wgpu::ShaderModuleDescriptor md{};md.nextInChain=&source;md.label="VR cockpit hands and wheel";
|
||||
auto shader=g_device.CreateShaderModule(&md);
|
||||
const wgpu::VertexAttribute attrs[]={{.format=wgpu::VertexFormat::Float32x4,.offset=0,.shaderLocation=0},
|
||||
{.format=wgpu::VertexFormat::Float32x3,.offset=16,.shaderLocation=1}};
|
||||
const wgpu::VertexBufferLayout layout{.arrayStride=28,.attributeCount=2,.attributes=attrs};
|
||||
const wgpu::ColorTargetState color{.format=format};
|
||||
const wgpu::FragmentState fragment{.module=shader,.entryPoint="fs",.targetCount=1,.targets=&color};
|
||||
const bool stencil=target.depthFormat==wgpu::TextureFormat::Depth24PlusStencil8;
|
||||
const wgpu::StencilFaceState mark{.compare=wgpu::CompareFunction::Always,
|
||||
.passOp=stencil?wgpu::StencilOperation::Replace:wgpu::StencilOperation::Keep};
|
||||
const wgpu::DepthStencilState depth{.format=target.depthFormat,.depthWriteEnabled=true,
|
||||
.depthCompare=reversedDepth?wgpu::CompareFunction::GreaterEqual:wgpu::CompareFunction::LessEqual,
|
||||
.stencilFront=mark,.stencilBack=mark,.stencilReadMask=1,.stencilWriteMask=stencil?1u:0u};
|
||||
wgpu::RenderPipelineDescriptor desc{};desc.label="VR cockpit";
|
||||
desc.vertex={.module=shader,.entryPoint="vs",.bufferCount=1,.buffers=&layout};
|
||||
desc.fragment=&fragment;desc.depthStencil=&depth;desc.multisample.count=target.msaaSamples;
|
||||
desc.primitive.topology=wgpu::PrimitiveTopology::TriangleList;
|
||||
pipeline=g_device.CreateRenderPipeline(&desc);pipelineSamples=target.msaaSamples;pipelineFormat=format;
|
||||
pipelineReversedDepth=reversedDepth;pipelineDepthFormat=target.depthFormat;
|
||||
}
|
||||
const auto revision=meshRevision.load();
|
||||
if(cachedMeshRevision!=revision || std::memcmp(&cachedCockpit,&frame.cockpit,sizeof(AuroraCockpit))!=0) {
|
||||
build_geometry(frame.cockpit,frameVertices);
|
||||
cachedCockpit=frame.cockpit;cachedMeshRevision=revision;
|
||||
}
|
||||
const auto& vertices=frameVertices;
|
||||
if(vertices.empty()) return;
|
||||
struct ClipVertex { float p[4]; V color; };
|
||||
static std::vector<ClipVertex> clip;
|
||||
clip.resize(vertices.size());
|
||||
const auto& projection=frame.eyes[eye].projection;
|
||||
for(size_t i=0;i<clip.size();++i) {
|
||||
const auto p=point(frame.cockpit.eyeFromSeat[eye],vertices[i].position);
|
||||
// The original race near plane can sit beyond a close hand. Keep that
|
||||
// hand at the nearest representable depth instead of clipping it away.
|
||||
const float z=sceneDepth.z*p[2]+sceneDepth.constant/std::max(frame.cockpit.unitsPerMeter,0.001f);
|
||||
clip[i]={{projection.m0[0]*p[0]+projection.m0[2]*p[2],projection.m1[1]*p[1]+projection.m1[2]*p[2],
|
||||
std::clamp(z,0.0f,std::max(-p[2],0.0f)),-p[2]},vertices[i].color};
|
||||
}
|
||||
const uint64_t bytes=clip.size()*sizeof(ClipVertex);
|
||||
if (!vertexBuffers[eye] || vertexCapacity[eye]<bytes) {
|
||||
vertexCapacity[eye]=(bytes+65535)&~uint64_t(65535);
|
||||
const wgpu::BufferDescriptor bd{.label="VR cockpit vertices",.usage=wgpu::BufferUsage::Vertex|wgpu::BufferUsage::CopyDst,
|
||||
.size=vertexCapacity[eye]};
|
||||
vertexBuffers[eye]=g_device.CreateBuffer(&bd);
|
||||
}
|
||||
auto& buffer=vertexBuffers[eye];
|
||||
g_queue.WriteBuffer(buffer,0,clip.data(),bytes);
|
||||
const wgpu::RenderPassColorAttachment attachment{.view=target.colorView,.resolveTarget=target.resolveView,
|
||||
.loadOp=wgpu::LoadOp::Load,.storeOp=wgpu::StoreOp::Store};
|
||||
const wgpu::RenderPassDepthStencilAttachment depth{.view=target.depthView,.depthLoadOp=wgpu::LoadOp::Load,
|
||||
.depthStoreOp=wgpu::StoreOp::Store,.depthClearValue=1.0f,
|
||||
.stencilLoadOp=target.depthFormat==wgpu::TextureFormat::Depth24PlusStencil8?wgpu::LoadOp::Load:wgpu::LoadOp::Undefined,
|
||||
.stencilStoreOp=target.depthFormat==wgpu::TextureFormat::Depth24PlusStencil8?wgpu::StoreOp::Store:wgpu::StoreOp::Undefined};
|
||||
const wgpu::RenderPassDescriptor pd{.label="VR cockpit overlay",.colorAttachmentCount=1,.colorAttachments=&attachment,.depthStencilAttachment=&depth};
|
||||
auto pass=existingPass?*existingPass:cmd.BeginRenderPass(&pd);
|
||||
pass.SetViewport(0,0,float(target.size.width),float(target.size.height),0,1);
|
||||
pass.SetScissorRect(0,0,target.size.width,target.size.height);
|
||||
// Mark only depth-visible samples; later virtual-screen draws test for zero.
|
||||
pass.SetStencilReference(1);
|
||||
pass.SetPipeline(pipeline);pass.SetVertexBuffer(0,buffer);pass.Draw(clip.size());
|
||||
pass.SetStencilReference(0);
|
||||
if(!existingPass) pass.End();
|
||||
}
|
||||
} // namespace aurora::gfx::cockpit
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "../gx/pipeline.hpp"
|
||||
#include "pipeline_cache.hpp"
|
||||
#include "stereo_replay.hpp"
|
||||
#include "cockpit.hpp"
|
||||
#include "tex_copy_conv.hpp"
|
||||
#include "tex_palette_conv.hpp"
|
||||
#include "texture_replacement.hpp"
|
||||
@@ -158,6 +159,9 @@ uint32_t g_mergedDrawCallCount = 0;
|
||||
|
||||
using CommandList = std::vector<Command>;
|
||||
struct RenderPass {
|
||||
// The world depth mapping of this pass's last full-view perspective draw, for
|
||||
// the VR cockpit overlay (set by prepare_stereo_replay_uniforms).
|
||||
cockpit::SceneDepth cockpitDepth{};
|
||||
wgpu::TextureView colorView;
|
||||
wgpu::TextureView resolveView; // MSAA resolve target; null if msaaSamples == 1
|
||||
wgpu::TextureView depthView;
|
||||
@@ -733,6 +737,13 @@ void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool cl
|
||||
.clearAlpha = clearAlpha,
|
||||
.clearDepth = clearDepth,
|
||||
}),
|
||||
.stereoPipeline = aurora::stereo_frame_provider_active() ? pipeline_ref(clear::PipelineConfig{
|
||||
.msaaSamples = msaaSamples,
|
||||
.clearColor = clearColor,
|
||||
.clearAlpha = clearAlpha,
|
||||
.clearDepth = clearDepth,
|
||||
.stereoStencil = true,
|
||||
}) : 0,
|
||||
.color =
|
||||
wgpu::Color{
|
||||
.r = clearColorValue.x(),
|
||||
@@ -1057,6 +1068,7 @@ void initialize() {
|
||||
}
|
||||
|
||||
void shutdown() {
|
||||
cockpit::shutdown();
|
||||
shutdown_pipeline_cache();
|
||||
gx::clear_shader_module_cache();
|
||||
efb_ram::shutdown();
|
||||
@@ -1512,6 +1524,7 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame,
|
||||
std::array<uint8_t, gx::MaxUniformSize> sourceUniform;
|
||||
std::array<uint8_t, gx::MaxUniformSize> eyeUniform;
|
||||
for (auto& pass : g_renderPasses) {
|
||||
pass.cockpitDepth = {};
|
||||
if (!pass.efbTarget) {
|
||||
continue;
|
||||
}
|
||||
@@ -1541,6 +1554,19 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame,
|
||||
std::memcpy(sourceUniform.data(), g_uniforms.data() + draw.uniformRange.offset, draw.uniformRange.size);
|
||||
Mat4x4<float> gameProjection;
|
||||
std::memcpy(&gameProjection, sourceUniform.data() + layout.projectionOffset, sizeof(gameProjection));
|
||||
// The VR cockpit overlay (hands, synthetic wheel) is drawn in metres and
|
||||
// depth-tested against the world, so it needs the world's own depth
|
||||
// mapping: the backend depth row of a full-view world draw, with this
|
||||
// viewport's depth range folded in because the overlay draws with 0..1.
|
||||
// Camera-attached effects share the camera's projection, so any full-view
|
||||
// perspective draw describes the same mapping.
|
||||
if (layout.perspective && !layout.nativeEfbEffect && gameProjection.m2[3] != 0.0f &&
|
||||
drawViewport.width >= displayRegion.width * 0.9f && drawViewport.height >= displayRegion.height * 0.9f) {
|
||||
const auto row = stereo_replay::backend_ndc_depth_row(gameProjection);
|
||||
const float low = std::clamp(std::min(drawViewport.znear, drawViewport.zfar), 0.f, 1.f);
|
||||
const float high = std::clamp(std::max(drawViewport.znear, drawViewport.zfar), 0.f, 1.f);
|
||||
pass.cockpitDepth = {row[2] * (high - low) - low, row[3] * (high - low), true};
|
||||
}
|
||||
// Only a genuinely affine projection carries its NDC position in its clip
|
||||
// position, which is what the virtual screen reprojection consumes. GX
|
||||
// tracks the projection type separately from the matrix, so a 2D draw
|
||||
@@ -1724,12 +1750,22 @@ struct RenderInvocation {
|
||||
uint32_t localPlayerCount = 1;
|
||||
// Inclusive index of the last pass to replay; -1 replays every pass.
|
||||
int32_t replayLastPass = -1;
|
||||
// Inclusive index of the last pass that does render work; texture bakes still run for the
|
||||
// passes after it. See last_pass_feeding_replay.
|
||||
int32_t renderLastPass = INT32_MAX;
|
||||
bool finalize = true;
|
||||
bool replayOnlyEfb = false;
|
||||
bool skipCopyClears = false;
|
||||
bool encodeTextureBakes = true;
|
||||
bool encodeResolves = true;
|
||||
bool captureDepth = true;
|
||||
// VR cockpit overlay, drawn inside the scene's pass just before the first
|
||||
// virtual-screen draw so the 2D layer's depth cannot hide it (see render_stereo_eye).
|
||||
const StereoReplayFrame* cockpitFrame = nullptr;
|
||||
wgpu::CommandEncoder* cockpitEncoder = nullptr;
|
||||
cockpit::SceneDepth cockpitDepth{};
|
||||
bool* cockpitDrawn = nullptr;
|
||||
bool* sceneDrawn = nullptr;
|
||||
};
|
||||
|
||||
static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vector<RenderPass>& passes, u32 idx,
|
||||
@@ -1740,6 +1776,9 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
|
||||
ZoneScoped;
|
||||
// Palette conversions, MSAA resolves and EFB copies depend on sealed frame state, not on the
|
||||
// interpolation weight, so encode them on the native render and let replay slots sample them.
|
||||
// Eye textures are reused; discard the previous frame's mask, then retain it
|
||||
// across guest passes even if the HUD clears or replaces guest depth.
|
||||
bool stencilInitialized = false;
|
||||
for (u32 i = 0; i < renderPasses.size(); ++i) {
|
||||
const auto& passInfo = renderPasses[i];
|
||||
if (invocation.replayLastPass >= 0 && i > static_cast<u32>(invocation.replayLastPass)) {
|
||||
@@ -1755,6 +1794,11 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
|
||||
tex_palette_conv::run(cmd, conv);
|
||||
}
|
||||
}
|
||||
if (static_cast<int32_t>(i) > invocation.renderLastPass) {
|
||||
// Nothing after the last replay-feeding resolve is shown or sampled on a headset; the
|
||||
// bakes above are all these passes owe the eye replays.
|
||||
continue;
|
||||
}
|
||||
const bool hasRenderWork = passInfo.clearColor || passInfo.clearDepth || !passInfo.commands.empty();
|
||||
if (i == renderPasses.size() - 1) {
|
||||
ASSERT(!passInfo.resolveTarget, "Final render pass must not have resolve target");
|
||||
@@ -1787,19 +1831,30 @@ static void render_impl(std::vector<RenderPass>& renderPasses, wgpu::CommandEnco
|
||||
},
|
||||
},
|
||||
};
|
||||
const bool stereoStencil = overrideTarget &&
|
||||
invocation.target->depthFormat == wgpu::TextureFormat::Depth24PlusStencil8;
|
||||
const wgpu::RenderPassDepthStencilAttachment depthStencilAttachment{
|
||||
.view = depthView,
|
||||
.depthLoadOp = passInfo.clearDepth && !dropCopyClear ? wgpu::LoadOp::Clear : wgpu::LoadOp::Load,
|
||||
.depthStoreOp = wgpu::StoreOp::Store,
|
||||
.depthClearValue = passInfo.clearDepthValue,
|
||||
.stencilLoadOp = stereoStencil ? (stencilInitialized ? wgpu::LoadOp::Load : wgpu::LoadOp::Clear) : wgpu::LoadOp::Undefined,
|
||||
.stencilStoreOp = stereoStencil ? wgpu::StoreOp::Store : wgpu::StoreOp::Undefined,
|
||||
.stencilClearValue = 0,
|
||||
};
|
||||
const GpuTimingCategory timingCategory = invocation.stereoEye == 0 ? GpuTimingCategory::EyeLeft
|
||||
: invocation.stereoEye == 1 ? GpuTimingCategory::EyeRight
|
||||
: invocation.interpolatedFrame >= 0 ? GpuTimingCategory::Interpolated
|
||||
: GpuTimingCategory::Mono;
|
||||
const wgpu::RenderPassDescriptor renderPassDescriptor{
|
||||
.label = render_pass_label(i),
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.depthStencilAttachment = &depthStencilAttachment,
|
||||
.timestampWrites = gpu_timing_pass(timingCategory),
|
||||
};
|
||||
|
||||
if (stereoStencil) stencilInitialized = true;
|
||||
auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
|
||||
render_pass_impl(pass, renderPasses, i, invocation);
|
||||
pass.End();
|
||||
@@ -1908,15 +1963,28 @@ void seal_frame(SealedFrame& out) noexcept {
|
||||
g_currentRenderPass = UINT32_MAX;
|
||||
}
|
||||
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize,
|
||||
int32_t nativeRenderLastPass) {
|
||||
render_impl(frame.data().passes, cmd,
|
||||
RenderInvocation{
|
||||
.interpolatedFrame = interpolatedFrame,
|
||||
.renderLastPass = nativeRenderLastPass,
|
||||
.finalize = finalize,
|
||||
.encodeTextureBakes = interpolatedFrame < 0,
|
||||
});
|
||||
}
|
||||
|
||||
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept {
|
||||
const auto& passes = frame.data().passes;
|
||||
int32_t last = -1;
|
||||
for (size_t i = 0; i < passes.size(); ++i) {
|
||||
if (passes[i].resolveTarget && !passes[i].displayCopyResolve) {
|
||||
last = static_cast<int32_t>(i);
|
||||
}
|
||||
}
|
||||
return last;
|
||||
}
|
||||
|
||||
bool has_late_stereo_replay(const SealedFrame& frame) noexcept {
|
||||
const auto& data = frame.data().stereo;
|
||||
return data.generation != 0 && data.generation == g_replayBufferGeneration.load(std::memory_order_acquire) &&
|
||||
@@ -1999,6 +2067,18 @@ void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, const Ster
|
||||
// The eye is a fresh per-frame attachment, not the reused EFB, so replaying
|
||||
// past that copy blanks the very image the game presented.
|
||||
const int32_t lastPass = get_stereo_stop_at_display_copy() ? displaySource.lastDisplayCopyPass : -1;
|
||||
cockpit::SceneDepth cockpitDepth{};
|
||||
for (size_t i = 0; i < frame.data().passes.size(); ++i) {
|
||||
if (lastPass >= 0 && i > static_cast<size_t>(lastPass)) {
|
||||
break;
|
||||
}
|
||||
if (frame.data().passes[i].cockpitDepth.valid) {
|
||||
cockpitDepth = frame.data().passes[i].cockpitDepth;
|
||||
}
|
||||
}
|
||||
bool cockpitDrawn = false;
|
||||
bool sceneDrawn = false;
|
||||
const bool cockpitActive = stereoFrame.cockpit.active && cockpitDepth.valid;
|
||||
render_impl(frame.data().passes, cmd,
|
||||
RenderInvocation{
|
||||
.stereoEye = eye,
|
||||
@@ -2012,7 +2092,17 @@ void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, const Ster
|
||||
.encodeTextureBakes = false,
|
||||
.encodeResolves = false,
|
||||
.captureDepth = false,
|
||||
.cockpitFrame = cockpitActive ? &stereoFrame : nullptr,
|
||||
.cockpitEncoder = &cmd,
|
||||
.cockpitDepth = cockpitDepth,
|
||||
.cockpitDrawn = &cockpitDrawn,
|
||||
.sceneDrawn = &sceneDrawn,
|
||||
});
|
||||
// A frame without a virtual-screen draw after its world still gets the
|
||||
// overlay, in a pass of its own over the finished eye.
|
||||
if (cockpitActive && !cockpitDrawn) {
|
||||
cockpit::render(cmd, stereoFrame, eye, cockpitDepth);
|
||||
}
|
||||
}
|
||||
|
||||
void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize) {
|
||||
@@ -2028,6 +2118,209 @@ void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame, bool finalize)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Per-pass GPU timing (see common.hpp) -------------------------------------------------------
|
||||
namespace {
|
||||
constexpr uint32_t kGpuTimingSlots = 4;
|
||||
constexpr uint32_t kGpuTimingPairs = 62;
|
||||
constexpr uint32_t kGpuTimingQueries = 2 * kGpuTimingPairs;
|
||||
|
||||
struct GpuTimingSlot {
|
||||
wgpu::QuerySet querySet;
|
||||
wgpu::Buffer resolve;
|
||||
wgpu::Buffer readback;
|
||||
std::array<wgpu::PassTimestampWrites, kGpuTimingPairs> writes{};
|
||||
std::array<GpuTimingCategory, kGpuTimingPairs> categories{};
|
||||
uint32_t pairs = 0;
|
||||
bool open = false; // between the frame's begin and end
|
||||
bool reading = false; // readback in flight or mapped
|
||||
bool mapped = false; // the callback ran; the encoding thread unmaps on reuse
|
||||
};
|
||||
|
||||
std::atomic<bool> g_gpuTimingEnabled{false};
|
||||
std::array<GpuTimingSlot, kGpuTimingSlots> g_gpuTimingSlots;
|
||||
uint32_t g_gpuTimingNextSlot = 0;
|
||||
int32_t g_gpuTimingCurrent = -1;
|
||||
bool g_gpuTimingReady = false;
|
||||
// Guards the totals below and every slot's reading/mapped flags: the map callback may run on
|
||||
// whichever thread processes Dawn's events.
|
||||
std::mutex g_gpuTimingMutex;
|
||||
std::array<uint64_t, static_cast<size_t>(GpuTimingCategory::Count)> g_gpuTimingTotalsNs{};
|
||||
uint64_t g_gpuTimingSpanNs = 0;
|
||||
uint32_t g_gpuTimingFrames = 0;
|
||||
uint32_t g_gpuTimingSkipped = 0;
|
||||
|
||||
bool gpu_timing_create_slots() {
|
||||
if (g_gpuTimingReady) {
|
||||
return true;
|
||||
}
|
||||
if (!webgpu::g_timestampQueriesSupported || !webgpu::g_device) {
|
||||
return false;
|
||||
}
|
||||
for (auto& slot : g_gpuTimingSlots) {
|
||||
const wgpu::QuerySetDescriptor querySetDescriptor{
|
||||
.label = "GPU timing queries",
|
||||
.type = wgpu::QueryType::Timestamp,
|
||||
.count = kGpuTimingQueries,
|
||||
};
|
||||
slot.querySet = webgpu::g_device.CreateQuerySet(&querySetDescriptor);
|
||||
const wgpu::BufferDescriptor resolveDescriptor{
|
||||
.label = "GPU timing resolve",
|
||||
.usage = wgpu::BufferUsage::QueryResolve | wgpu::BufferUsage::CopySrc,
|
||||
.size = kGpuTimingQueries * sizeof(uint64_t),
|
||||
};
|
||||
slot.resolve = webgpu::g_device.CreateBuffer(&resolveDescriptor);
|
||||
const wgpu::BufferDescriptor readbackDescriptor{
|
||||
.label = "GPU timing readback",
|
||||
.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst,
|
||||
.size = kGpuTimingQueries * sizeof(uint64_t),
|
||||
};
|
||||
slot.readback = webgpu::g_device.CreateBuffer(&readbackDescriptor);
|
||||
}
|
||||
g_gpuTimingReady = true;
|
||||
return true;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void gpu_timing_set_enabled(bool enabled) noexcept { g_gpuTimingEnabled.store(enabled, std::memory_order_relaxed); }
|
||||
bool gpu_timing_enabled() noexcept { return g_gpuTimingEnabled.load(std::memory_order_relaxed); }
|
||||
|
||||
void gpu_timing_begin_frame() noexcept {
|
||||
g_gpuTimingCurrent = -1;
|
||||
if (!gpu_timing_enabled() || !gpu_timing_create_slots()) {
|
||||
return;
|
||||
}
|
||||
const uint32_t index = g_gpuTimingNextSlot;
|
||||
g_gpuTimingNextSlot = (g_gpuTimingNextSlot + 1) % kGpuTimingSlots;
|
||||
auto& slot = g_gpuTimingSlots[index];
|
||||
{
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
if (slot.reading && !slot.mapped) {
|
||||
++g_gpuTimingSkipped; // the GPU is more than a ring behind; leave this frame untimed
|
||||
return;
|
||||
}
|
||||
if (slot.mapped) {
|
||||
slot.readback.Unmap();
|
||||
slot.mapped = false;
|
||||
}
|
||||
slot.reading = false;
|
||||
}
|
||||
slot.pairs = 0;
|
||||
slot.open = true;
|
||||
g_gpuTimingCurrent = static_cast<int32_t>(index);
|
||||
}
|
||||
|
||||
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept {
|
||||
if (g_gpuTimingCurrent < 0) {
|
||||
return nullptr;
|
||||
}
|
||||
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
|
||||
if (!slot.open || slot.pairs >= kGpuTimingPairs) {
|
||||
return nullptr;
|
||||
}
|
||||
const uint32_t i = slot.pairs++;
|
||||
slot.writes[i] = wgpu::PassTimestampWrites{
|
||||
.querySet = slot.querySet,
|
||||
.beginningOfPassWriteIndex = 2 * i,
|
||||
.endOfPassWriteIndex = 2 * i + 1,
|
||||
};
|
||||
slot.categories[i] = category;
|
||||
return &slot.writes[i];
|
||||
}
|
||||
|
||||
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept {
|
||||
if (g_gpuTimingCurrent < 0) {
|
||||
return;
|
||||
}
|
||||
auto& slot = g_gpuTimingSlots[static_cast<size_t>(g_gpuTimingCurrent)];
|
||||
slot.open = false;
|
||||
if (slot.pairs == 0) {
|
||||
g_gpuTimingCurrent = -1;
|
||||
return;
|
||||
}
|
||||
const uint32_t queries = 2 * slot.pairs;
|
||||
encoder.ResolveQuerySet(slot.querySet, 0, queries, slot.resolve, 0);
|
||||
encoder.CopyBufferToBuffer(slot.resolve, 0, slot.readback, 0, queries * sizeof(uint64_t));
|
||||
}
|
||||
|
||||
void gpu_timing_after_submit() noexcept {
|
||||
if (g_gpuTimingCurrent < 0) {
|
||||
return;
|
||||
}
|
||||
const uint32_t index = static_cast<uint32_t>(g_gpuTimingCurrent);
|
||||
g_gpuTimingCurrent = -1;
|
||||
auto& slot = g_gpuTimingSlots[index];
|
||||
const uint32_t pairs = slot.pairs;
|
||||
{
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
slot.reading = true;
|
||||
slot.mapped = false;
|
||||
}
|
||||
slot.readback.MapAsync(
|
||||
wgpu::MapMode::Read, 0, 2 * pairs * sizeof(uint64_t), wgpu::CallbackMode::AllowSpontaneous,
|
||||
[index, pairs](wgpu::MapAsyncStatus status, wgpu::StringView) {
|
||||
auto& slot = g_gpuTimingSlots[index];
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
if (status != wgpu::MapAsyncStatus::Success) {
|
||||
slot.reading = false;
|
||||
return;
|
||||
}
|
||||
const auto* stamps =
|
||||
static_cast<const uint64_t*>(slot.readback.GetConstMappedRange(0, 2 * pairs * sizeof(uint64_t)));
|
||||
if (stamps != nullptr) {
|
||||
uint64_t first = UINT64_MAX;
|
||||
uint64_t last = 0;
|
||||
for (uint32_t i = 0; i < pairs; ++i) {
|
||||
const uint64_t begin = stamps[2 * i];
|
||||
const uint64_t end = stamps[2 * i + 1];
|
||||
if (end < begin) {
|
||||
continue;
|
||||
}
|
||||
g_gpuTimingTotalsNs[static_cast<size_t>(slot.categories[i])] += end - begin;
|
||||
first = std::min(first, begin);
|
||||
last = std::max(last, end);
|
||||
}
|
||||
if (last > first) {
|
||||
g_gpuTimingSpanNs += last - first;
|
||||
}
|
||||
++g_gpuTimingFrames;
|
||||
}
|
||||
slot.mapped = true;
|
||||
});
|
||||
}
|
||||
|
||||
std::string gpu_timing_report() {
|
||||
std::lock_guard lock(g_gpuTimingMutex);
|
||||
if (g_gpuTimingFrames == 0 && g_gpuTimingSkipped == 0) {
|
||||
return {};
|
||||
}
|
||||
static constexpr std::array<const char*, static_cast<size_t>(GpuTimingCategory::Count)> kNames{
|
||||
"mono", "eyeL", "eyeR", "interp", "screen", "panel", "efbcopy", "palette", "peek", "snapshot", "present"};
|
||||
std::string text;
|
||||
if (g_gpuTimingFrames != 0) {
|
||||
const double frames = g_gpuTimingFrames;
|
||||
uint64_t sum = 0;
|
||||
text += fmt::format("GPU ms/frame over {} frames: passes-span={:.2f}", g_gpuTimingFrames,
|
||||
static_cast<double>(g_gpuTimingSpanNs) / 1e6 / frames);
|
||||
for (size_t i = 0; i < kNames.size(); ++i) {
|
||||
if (g_gpuTimingTotalsNs[i] == 0) {
|
||||
continue;
|
||||
}
|
||||
sum += g_gpuTimingTotalsNs[i];
|
||||
text += fmt::format(" {}={:.2f}", kNames[i], static_cast<double>(g_gpuTimingTotalsNs[i]) / 1e6 / frames);
|
||||
}
|
||||
const uint64_t between = g_gpuTimingSpanNs > sum ? g_gpuTimingSpanNs - sum : 0;
|
||||
text += fmt::format(" between-passes={:.2f}", static_cast<double>(between) / 1e6 / frames);
|
||||
}
|
||||
if (g_gpuTimingSkipped != 0) {
|
||||
text += fmt::format(" (untimed frames: {})", g_gpuTimingSkipped);
|
||||
}
|
||||
g_gpuTimingTotalsNs.fill(0);
|
||||
g_gpuTimingSpanNs = 0;
|
||||
g_gpuTimingFrames = 0;
|
||||
g_gpuTimingSkipped = 0;
|
||||
return text;
|
||||
}
|
||||
|
||||
void after_submit() noexcept {
|
||||
depth_peek::after_submit();
|
||||
efb_ram::after_submit();
|
||||
@@ -2223,6 +2516,24 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
|
||||
draw.gx.interpolatedUniformRanges[invocation.interpolatedFrame].size != 0) {
|
||||
uniformOverride = &draw.gx.interpolatedUniformRanges[invocation.interpolatedFrame];
|
||||
}
|
||||
// Draw against world depth and mark visible cockpit samples before HUD
|
||||
// depth replaces it. The screen pipelines reject those stencil samples.
|
||||
if (invocation.cockpitFrame != nullptr && overrideTarget) {
|
||||
if (draw.gx.uniformReplayLayout.perspective) {
|
||||
*invocation.sceneDrawn = true;
|
||||
}
|
||||
if (virtualScreenDraw && *invocation.sceneDrawn && !*invocation.cockpitDrawn) {
|
||||
cockpit::render(*invocation.cockpitEncoder, *invocation.cockpitFrame, invocation.stereoEye,
|
||||
invocation.cockpitDepth, &pass);
|
||||
*invocation.cockpitDrawn = true;
|
||||
encodeState = {};
|
||||
encodeState.boundTextureBindGroup = gx::g_emptyTextureBindGroup.Get();
|
||||
pass.SetBindGroup(0, g_staticBindGroup);
|
||||
pass.SetBindGroup(2, gx::g_emptyTextureBindGroup);
|
||||
scissorStateKnown = false;
|
||||
viewportStateKnown = false;
|
||||
}
|
||||
}
|
||||
// Such a draw no longer lands where the game aimed it, while the
|
||||
// recorded scissor still describes the rectangle it occupied on the flat
|
||||
// frame (Mario Kart clips the item roulette that way). Honouring that
|
||||
@@ -2240,10 +2551,15 @@ static void render_pass_impl(const wgpu::RenderPassEncoder& pass, const std::vec
|
||||
apply_viewport(fullEyeDraw);
|
||||
}
|
||||
gx::render(draw.gx, pass, encodeState, renderPasses[idx].requireReadyPipelines, uniformOverride,
|
||||
virtualScreenDraw ? draw.gx.exactScreenDepthPipeline : 0);
|
||||
overrideTarget && invocation.target->depthFormat == wgpu::TextureFormat::Depth24PlusStencil8
|
||||
? (virtualScreenDraw ? draw.gx.stereoScreenPipeline : draw.gx.stereoPipeline)
|
||||
: (virtualScreenDraw ? draw.gx.exactScreenDepthPipeline : 0));
|
||||
} break;
|
||||
case ShaderType::Clear: {
|
||||
auto clearDraw = draw.clear;
|
||||
if (overrideTarget && invocation.target->depthFormat == wgpu::TextureFormat::Depth24PlusStencil8) {
|
||||
clearDraw.pipeline = clearDraw.stereoPipeline;
|
||||
}
|
||||
if (multiplayer) {
|
||||
const auto& sc = clearDraw.scissor;
|
||||
if (clearDraw.copyClear ||
|
||||
@@ -2478,3 +2794,32 @@ void aurora_pop_debug_group() {
|
||||
}
|
||||
|
||||
const AuroraStats* aurora_get_stats() { return &aurora::gfx::g_stats; }
|
||||
|
||||
void aurora_set_vr_hand_mesh(uint32_t hand, const AuroraVRHandVertex* vertices, uint32_t vertexCount,
|
||||
const uint16_t* indices, uint32_t indexCount, const float* bindPoses,
|
||||
const int32_t* parents, uint32_t jointCount) {
|
||||
using namespace aurora::gfx::cockpit;
|
||||
if (hand >= 2) {
|
||||
return;
|
||||
}
|
||||
std::shared_ptr<HandMesh> mesh;
|
||||
if (vertices && indices && bindPoses && parents && jointCount == 26 && vertexCount > 0 && vertexCount <= 65535 &&
|
||||
indexCount <= 100000 && indexCount % 3 == 0) {
|
||||
for (uint32_t i = 0; i < indexCount; ++i) {
|
||||
if (indices[i] >= vertexCount) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
mesh = std::make_shared<HandMesh>();
|
||||
mesh->vertices.assign(vertices, vertices + vertexCount);
|
||||
mesh->indices.assign(indices, indices + indexCount);
|
||||
for (int j = 0; j < 26; ++j) {
|
||||
mesh->bind[j] = from_pose(bindPoses + j * 7);
|
||||
mesh->inverseBind[j] = inverse(mesh->bind[j]);
|
||||
mesh->parents[j] = parents[j];
|
||||
}
|
||||
}
|
||||
std::lock_guard lock(meshMutex);
|
||||
meshes[hand] = std::move(mesh);
|
||||
++meshRevision;
|
||||
}
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <cstring>
|
||||
#include <array>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
@@ -290,6 +291,7 @@ struct ReplayTarget {
|
||||
wgpu::TextureView copySourceDepthView;
|
||||
wgpu::Extent3D size{};
|
||||
uint32_t msaaSamples = 1;
|
||||
wgpu::TextureFormat depthFormat = wgpu::TextureFormat::Depth32Float;
|
||||
};
|
||||
|
||||
struct StereoReplayEye {
|
||||
@@ -313,6 +315,8 @@ struct StereoReplayEye {
|
||||
|
||||
struct StereoReplayFrame {
|
||||
std::array<StereoReplayEye, AURORA_STEREO_EYE_COUNT> eyes;
|
||||
// VR hands and synthetic wheel, drawn per eye after the world (gfx/cockpit.hpp).
|
||||
AuroraCockpit cockpit{};
|
||||
};
|
||||
|
||||
void end_frame(const wgpu::CommandEncoder& cmd);
|
||||
@@ -358,7 +362,14 @@ bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, c
|
||||
|
||||
// Encode a sealed frame. Never touches the producer-visible recording state,
|
||||
// so this may run concurrently with the producer's FIFO drains.
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true);
|
||||
// `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every
|
||||
// pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it
|
||||
// after the last pass whose EFB copy the eye replays sample.
|
||||
void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true,
|
||||
int32_t nativeRenderLastPass = INT32_MAX);
|
||||
// Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1
|
||||
// when no pass does: everything after it exists only for the presented image.
|
||||
int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept;
|
||||
|
||||
// Replays only main-EFB passes into one Aurora-owned eye target. Native
|
||||
// offscreen/EFB-copy passes are consumed from the mono render and are not
|
||||
@@ -414,6 +425,39 @@ bool is_offscreen() noexcept;
|
||||
uint32_t get_sample_count() noexcept;
|
||||
void clear_caches() noexcept;
|
||||
|
||||
// Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery,
|
||||
// every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the
|
||||
// frame's queries are resolved into a small ring of readback buffers and the completed frames'
|
||||
// durations are summed per category until gpu_timing_report() consumes them. Off by default:
|
||||
// aurora.cpp enables it together with the Android frame-rate log.
|
||||
enum class GpuTimingCategory : uint8_t {
|
||||
Mono, // the native (desktop) render of the recorded GX passes
|
||||
EyeLeft, // stereo replay of the left eye
|
||||
EyeRight, // stereo replay of the right eye
|
||||
Interpolated, // interpolated presentation slots
|
||||
VirtualScreen, // the 2D virtual screen built for each eye
|
||||
Panel, // the in-headset settings panel
|
||||
EfbCopy, // EFB copy format conversions
|
||||
Palette, // palette (TLUT) texture conversions
|
||||
DepthPeek, // the depth snapshot compute pass
|
||||
Snapshot, // presentation snapshot and its ImGui pass
|
||||
Present, // the desktop presentation copy
|
||||
Count,
|
||||
};
|
||||
void gpu_timing_set_enabled(bool enabled) noexcept;
|
||||
bool gpu_timing_enabled() noexcept;
|
||||
// Opens the current frame's query slot; a frame whose slot is still being read back is skipped.
|
||||
void gpu_timing_begin_frame() noexcept;
|
||||
// Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted.
|
||||
const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept;
|
||||
// Resolves the open frame's queries on `encoder`, which must be the frame's last submission.
|
||||
void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept;
|
||||
// After that submission: starts the readback of the resolved queries.
|
||||
void gpu_timing_after_submit() noexcept;
|
||||
// Per-frame averages of the frames read back since the last call, formatted for the log, or
|
||||
// an empty string when nothing was measured.
|
||||
std::string gpu_timing_report();
|
||||
|
||||
namespace tex_palette_conv {
|
||||
struct ConvRequest;
|
||||
} // namespace tex_palette_conv
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "depth_peek.hpp"
|
||||
#include "common.hpp"
|
||||
|
||||
#include "../dolphin/vi/vi_internal.hpp"
|
||||
#include "../gx/gx.hpp"
|
||||
@@ -403,6 +404,7 @@ void encode_frame_snapshot(const wgpu::CommandEncoder& cmd, const wgpu::TextureV
|
||||
|
||||
const wgpu::ComputePassDescriptor passDescriptor{
|
||||
.label = "Depth Peek Compute Pass",
|
||||
.timestampWrites = gpu_timing_pass(GpuTimingCategory::DepthPeek),
|
||||
};
|
||||
const auto pass = cmd.BeginComputePass(&passDescriptor);
|
||||
pass.SetPipeline(g_pipeline);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "tex_copy_conv.hpp"
|
||||
#include "common.hpp"
|
||||
#include "tex_copy_format_contract.hpp"
|
||||
|
||||
#include "../internal.hpp"
|
||||
@@ -651,6 +652,7 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
|
||||
.label = "TexCopyConv Pass",
|
||||
.colorAttachmentCount = colorAttachments.size(),
|
||||
.colorAttachments = colorAttachments.data(),
|
||||
.timestampWrites = gpu_timing_pass(GpuTimingCategory::EfbCopy),
|
||||
};
|
||||
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetPipeline(pipeline);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "tex_palette_conv.hpp"
|
||||
#include "common.hpp"
|
||||
|
||||
#include "../internal.hpp"
|
||||
#include "../webgpu/gpu.hpp"
|
||||
@@ -249,6 +250,7 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
|
||||
.label = "TexPaletteConv Pass",
|
||||
.colorAttachmentCount = colorAttachments.size(),
|
||||
.colorAttachments = colorAttachments.data(),
|
||||
.timestampWrites = gpu_timing_pass(GpuTimingCategory::Palette),
|
||||
};
|
||||
const auto pass = cmd.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetPipeline(pipeline);
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
#include "../gfx/texture_replacement.hpp"
|
||||
#include "dolphin/gx/GXAurora.h"
|
||||
#include "gx.hpp"
|
||||
#include "native_wheel.hpp"
|
||||
#include "gx_fmt.hpp"
|
||||
#include "pipeline.hpp"
|
||||
#include "shader_info.hpp"
|
||||
@@ -1929,7 +1930,8 @@ static u32 calculate_last_vtx_size(GXVtxFmt fmt) {
|
||||
|
||||
static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange,
|
||||
uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature,
|
||||
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride);
|
||||
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride,
|
||||
NativeWheelArray* nativeWheel);
|
||||
|
||||
// The per-draw geometry signature, matrix-usage mask and draw-identity hashes exist purely to feed frame interpolation
|
||||
// (build_uniform consumes them only after its `frame_interpolation_fps() == 0` early-out).
|
||||
@@ -1959,6 +1961,25 @@ static uint32_t matrix_index_prefix_size(GXVtxFmt fmt) noexcept {
|
||||
return size;
|
||||
}
|
||||
|
||||
// Which animated vertex array, if any, this draw takes (native_wheel.hpp). Called once per draw, before the merge
|
||||
// test, because a merged draw renders through the binding the draw it folds into resolved.
|
||||
static NativeWheelArray* resolve_native_wheel(GXVtxFmt fmt, const uint8_t* vertices, u16 vtxCount,
|
||||
uint32_t vtxStride) noexcept {
|
||||
// A direct-position draw reads no array at all, and g_gxState.arrays[GX_VA_POS] then still holds whatever was
|
||||
// bound last, which must not be matched against.
|
||||
if (g_gxState.vtxDesc[GX_VA_POS] != GX_INDEX8 && g_gxState.vtxDesc[GX_VA_POS] != GX_INDEX16)
|
||||
LIKELY { return nullptr; }
|
||||
const auto& array = g_gxState.arrays[GX_VA_POS];
|
||||
if (nativeWheelArrays.empty())
|
||||
LIKELY {
|
||||
if (!nativeWheelPreviousSources.empty())
|
||||
UNLIKELY { native_wheel_note_outside(array.data); }
|
||||
return nullptr;
|
||||
}
|
||||
return native_wheel_array(array, vertices, static_cast<u32>(vtxCount) * vtxStride, vtxStride,
|
||||
matrix_index_prefix_size(fmt));
|
||||
}
|
||||
|
||||
// Screen-space bounds of a simple orthographic rectangle or line, textured or
|
||||
// not: MKW's split-screen partition is a layout picture pane (a one-pixel quad
|
||||
// sampling a pattern texture), so texture use cannot disqualify a candidate.
|
||||
@@ -2185,6 +2206,10 @@ static ArrayRef<u16> offset_index_template(const CachedIndexTemplate& indexTempl
|
||||
|
||||
struct CachedPipelineState {
|
||||
gfx::PipelineRef ref = 0;
|
||||
const PipelineConfig* config = nullptr;
|
||||
mutable gfx::PipelineRef stereoRef = 0;
|
||||
mutable gfx::PipelineRef screenRef = 0;
|
||||
mutable gfx::PipelineRef stereoScreenRef = 0;
|
||||
HashType configHash = 0;
|
||||
// Carried here so the draw can be recorded without keeping the PipelineConfig that produced it alive; it is the only
|
||||
// field of the config the draw itself still needs.
|
||||
@@ -2210,6 +2235,7 @@ static const CachedPipelineState& cached_pipeline_state(const PipelineConfig& co
|
||||
entry.config = config;
|
||||
entry.state = {
|
||||
.ref = gfx::pipeline_ref(config),
|
||||
.config = &entry.config,
|
||||
.configHash = hash,
|
||||
.dstAlpha = config.dstAlpha,
|
||||
.shaderInfo = build_shader_info(config.shaderConfig),
|
||||
@@ -2252,37 +2278,20 @@ static const CachedPipelineState& resolve_pipeline_state(GXPrimitive prim, GXVtx
|
||||
return state;
|
||||
}
|
||||
|
||||
// Exact Screen Depth changes shader outputs but no GX pipeline state. Resolve a
|
||||
// sibling pipeline only for orthographic draws that can actually reach the VR
|
||||
// screen, and memoize it with the same state epoch as the ordinary pipeline.
|
||||
static gfx::PipelineRef resolve_exact_screen_depth_pipeline(GXPrimitive prim, GXVtxFmt fmt) {
|
||||
struct Memo {
|
||||
gfx::PipelineRef ref = 0;
|
||||
u32 generation = 0;
|
||||
u32 sampleCount = 0;
|
||||
GXPrimitive prim = static_cast<GXPrimitive>(0);
|
||||
GXVtxFmt fmt = static_cast<GXVtxFmt>(0);
|
||||
};
|
||||
static Memo memo{};
|
||||
|
||||
const u32 sampleCount = gfx::get_sample_count();
|
||||
const u32 generation = g_gxState.pipelineStateGeneration;
|
||||
if (memo.ref != 0 && memo.generation == generation && memo.sampleCount == sampleCount && memo.prim == prim &&
|
||||
memo.fmt == fmt)
|
||||
LIKELY { return memo.ref; }
|
||||
|
||||
PipelineConfig config{};
|
||||
populate_pipeline_config(config, prim, fmt);
|
||||
config.shaderConfig.exactScreenDepth = 1;
|
||||
const gfx::PipelineRef ref = gfx::pipeline_ref(config);
|
||||
memo = Memo{
|
||||
.ref = ref,
|
||||
.generation = generation,
|
||||
.sampleCount = sampleCount,
|
||||
.prim = prim,
|
||||
.fmt = fmt,
|
||||
};
|
||||
return ref;
|
||||
// Lazily cache eye-format siblings alongside the ordinary pipeline. Steady-state
|
||||
// draws only read the refs: no extra config population/hashing on the Quest CPU.
|
||||
// Shader modules are shared by the depth-format variants.
|
||||
static void resolve_replay_pipelines(const CachedPipelineState& state, bool screen) {
|
||||
if (state.stereoRef && (!screen || state.stereoScreenRef)) return;
|
||||
PipelineConfig config = *state.config;
|
||||
config.stereoStencil = 1;
|
||||
if (!state.stereoRef) state.stereoRef = gfx::pipeline_ref(config);
|
||||
if (screen && !state.stereoScreenRef) {
|
||||
config.shaderConfig.exactScreenDepth = 1;
|
||||
state.stereoScreenRef = gfx::pipeline_ref(config);
|
||||
config.stereoStencil = 0;
|
||||
state.screenRef = gfx::pipeline_ref(config);
|
||||
}
|
||||
}
|
||||
|
||||
bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, uint16_t vtxCount, uint32_t vertexBytes) {
|
||||
@@ -2321,7 +2330,8 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui
|
||||
const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{};
|
||||
handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature,
|
||||
interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0,
|
||||
interpolationIdentityActive, vertices, vtxSize);
|
||||
interpolationIdentityActive, vertices, vtxSize,
|
||||
resolve_native_wheel(fmt, vertices, vtxCount, vtxSize));
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -2354,13 +2364,21 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
|
||||
gfx::Range vertRange = push_draw_vertices(vertices, vtxCount, vtxSize);
|
||||
pos += totalVtxBytes;
|
||||
|
||||
// Try to merge with previous draw call
|
||||
// The animated vertex array this draw takes is decided per draw, and the decision is part of what a merge would
|
||||
// share, so resolve it here and hand the result to handle_draw_unmerged rather than deciding twice.
|
||||
NativeWheelArray* const nativeWheel = resolve_native_wheel(fmt, vertices, vtxCount, vtxSize);
|
||||
|
||||
// Try to merge with previous draw call.
|
||||
if (!g_gxState.stateDirty && !(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC))
|
||||
LIKELY {
|
||||
auto* lastDraw = gfx::get_last_draw_command<DrawData>();
|
||||
// Only if the previous draw call was a single instance draw (no lines/points handling)
|
||||
// Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw
|
||||
// that resolved the same animated array: the merged whole renders through that draw's binding. Anything the
|
||||
// decision cache cannot vouch for (a command it was not recorded against) stays unmerged.
|
||||
if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS &&
|
||||
lastDraw->instanceCount == 1)
|
||||
lastDraw->instanceCount == 1 &&
|
||||
(nativeWheelArrays.empty() ||
|
||||
(nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel)))
|
||||
LIKELY {
|
||||
const auto& indexTemplate = cached_index_template(prim, vtxCount);
|
||||
const auto indices = offset_index_template(indexTemplate, lastDraw->vtxCount);
|
||||
@@ -2389,13 +2407,14 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
|
||||
const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{};
|
||||
handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature,
|
||||
interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0,
|
||||
interpolationIdentityActive, vertices, vtxSize);
|
||||
interpolationIdentityActive, vertices, vtxSize, nativeWheel);
|
||||
return true;
|
||||
}
|
||||
|
||||
static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange,
|
||||
uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature,
|
||||
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride) {
|
||||
bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride,
|
||||
NativeWheelArray* nativeWheel) {
|
||||
ZoneScoped;
|
||||
// GX_CULL_ALL rasterizes nothing on hardware - no color, no depth.
|
||||
if (g_gxState.cullMode == GX_CULL_ALL && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS)
|
||||
@@ -2419,6 +2438,24 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
|
||||
}
|
||||
auto& array = g_gxState.arrays[i];
|
||||
const u32 uploadStride = padded_upload_stride(array.stride);
|
||||
if (i == GX_VA_POS && nativeWheel != nullptr)
|
||||
UNLIKELY {
|
||||
static unsigned nativeWheelDrawLogs = 0;
|
||||
if (nativeWheelDrawLogs++ < 4) Log.info("Native steering wheel: animated local vehicle vertex array");
|
||||
// Never populate the shared source's cache with the animated copy: later draws of the same asset must
|
||||
// still see the original vertices. The copy takes the same padded upload path as the original.
|
||||
if (nativeWheel->uploaded.size == 0 || nativeWheel->uploadedStride != uploadStride) {
|
||||
AttrArray animated{};
|
||||
animated.data = nativeWheel->bytes.data();
|
||||
animated.size = array.size;
|
||||
animated.stride = array.stride;
|
||||
animated.le = array.le;
|
||||
nativeWheel->uploaded = push_vertex_array(animated, uploadStride);
|
||||
nativeWheel->uploadedStride = uploadStride;
|
||||
}
|
||||
ranges.vaRanges[0] = nativeWheel->uploaded;
|
||||
continue;
|
||||
}
|
||||
if (array.cachedRange.size > 0 && array.cachedStride == uploadStride) {
|
||||
ranges.vaRanges[i - GX_VA_POS] = array.cachedRange;
|
||||
} else {
|
||||
@@ -2459,10 +2496,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
|
||||
const bool perspective = g_gxState.projType == GX_PERSPECTIVE;
|
||||
const auto uniformRanges = build_uniform(info, vertRange.offset, ranges, drawIdentity, perspective, usedPnMtxMask);
|
||||
const auto& replayLayout = uniformRanges.replayLayout;
|
||||
const gfx::PipelineRef exactScreenDepthPipeline =
|
||||
aurora::stereo_frame_provider_active() && !replayLayout.perspective && !replayLayout.nativeEfbEffect
|
||||
? resolve_exact_screen_depth_pipeline(prim, fmt)
|
||||
: 0;
|
||||
const bool stereo = aurora::stereo_frame_provider_active();
|
||||
const bool screen = !replayLayout.perspective && !replayLayout.nativeEfbEffect;
|
||||
if (stereo) resolve_replay_pipelines(pipelineState, screen);
|
||||
s_lastDrawRecordedInterpolation = interpolationIdentityActive;
|
||||
|
||||
uint32_t instanceCount = 1;
|
||||
@@ -2475,7 +2511,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
|
||||
}
|
||||
gfx::push_draw_command(DrawData{
|
||||
.pipeline = pipeline,
|
||||
.exactScreenDepthPipeline = exactScreenDepthPipeline,
|
||||
.exactScreenDepthPipeline = stereo && screen ? pipelineState.screenRef : 0,
|
||||
.stereoPipeline = stereo ? pipelineState.stereoRef : 0,
|
||||
.stereoScreenPipeline = stereo && screen ? pipelineState.stereoScreenRef : 0,
|
||||
.vertRange = vertRange,
|
||||
.idxRange = idxRange,
|
||||
.uniformRange = uniformRanges.current,
|
||||
@@ -2490,6 +2528,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
|
||||
.dstAlpha = pipelineState.dstAlpha,
|
||||
.screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride),
|
||||
});
|
||||
// What the next draw must match to be allowed to fold into this one.
|
||||
nativeWheelLastDrawCommand = gfx::get_last_draw_command<DrawData>();
|
||||
nativeWheelLastDecision = nativeWheel;
|
||||
g_gxState.stateDirty = false;
|
||||
}
|
||||
|
||||
|
||||
@@ -1609,10 +1609,18 @@ static inline wgpu::PrimitiveState to_primitive_state(GXCullMode gx_cullMode) {
|
||||
wgpu::RenderPipeline build_pipeline(const PipelineConfig& config, ArrayRef<wgpu::VertexBufferLayout> vtxBuffers,
|
||||
wgpu::ShaderModule shader, const char* label) noexcept {
|
||||
ZoneScoped;
|
||||
const bool maskCockpit = config.stereoStencil && config.shaderConfig.exactScreenDepth;
|
||||
const wgpu::StencilFaceState stencil{
|
||||
.compare = maskCockpit ? wgpu::CompareFunction::Equal : wgpu::CompareFunction::Always,
|
||||
};
|
||||
const wgpu::DepthStencilState depthStencil{
|
||||
.format = g_graphicsConfig.depthFormat,
|
||||
.format = config.stereoStencil ? wgpu::TextureFormat::Depth24PlusStencil8 : g_graphicsConfig.depthFormat,
|
||||
.depthWriteEnabled = config.depthUpdate,
|
||||
.depthCompare = config.depthCompare ? to_compare_function(config.depthFunc) : wgpu::CompareFunction::Always,
|
||||
.stencilFront = stencil,
|
||||
.stencilBack = stencil,
|
||||
.stencilReadMask = 1,
|
||||
.stencilWriteMask = 0,
|
||||
};
|
||||
const auto blendState = to_blend_state(config.blendMode, config.blendFacSrc, config.blendFacDst, config.blendOp,
|
||||
config.pixelFmt, config.dstAlpha);
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
|
||||
//
|
||||
// The VR first-person camera animates the local vehicle's steering wheel by
|
||||
// handing Aurora a rotated copy of one of the vehicle's position arrays. The
|
||||
// copy applies only to a draw that binds that exact array *and* carries the
|
||||
// local vehicle's model-view matrix, so an opponent sharing the asset keeps
|
||||
// the original vertices. Everything here runs on the GX (command processor)
|
||||
// thread; the set/clear entry points are posted there by the runtime.
|
||||
#pragma once
|
||||
#include "gx.hpp"
|
||||
#include <algorithm>
|
||||
#include <vector>
|
||||
#include <cstring>
|
||||
#include <cmath>
|
||||
#include <atomic>
|
||||
#include <aurora/native_wheel_match.hpp>
|
||||
namespace aurora::gx {
|
||||
struct NativeWheelArray {
|
||||
const void* source{};
|
||||
std::vector<uint8_t> bytes;
|
||||
std::array<float,12> modelView{};
|
||||
gfx::Range uploaded{};
|
||||
// Element stride of `uploaded`, which differs from the array's when the upload is padded.
|
||||
uint32_t uploadedStride=0;
|
||||
};
|
||||
inline std::vector<NativeWheelArray> nativeWheelArrays;
|
||||
inline uint32_t nativeWheelMatches=0;
|
||||
inline std::atomic<uint32_t> nativeWheelLastMatches{0};
|
||||
|
||||
// A draw folds into the previous one by appending its vertices to that draw's
|
||||
// range, so the merged whole renders through the *first* draw's array binding:
|
||||
// two draws may only merge when they resolved the same replacement. The
|
||||
// decision below is therefore taken once per draw (the ownership walk is far
|
||||
// too costly to repeat) and kept with the command it was recorded for. It is
|
||||
// cleared with the set, which the runtime posts once a frame, so a command
|
||||
// address a later frame's list reuses can never be read as a hit.
|
||||
inline NativeWheelArray* nativeWheelLastDecision=nullptr;
|
||||
inline const void* nativeWheelLastDrawCommand=nullptr;
|
||||
|
||||
// For the host log: why draws of the replaced arrays did or did not take them.
|
||||
struct NativeWheelDiagnostics {
|
||||
uint32_t sets=0; // replacement sets cleared since the last report
|
||||
uint32_t boundDraws=0; // draws that bound a replacement's source array
|
||||
uint32_t oversizeDraws=0; // ... whose bound range was larger than the replacement
|
||||
uint32_t outsideDraws=0; // draws that bound a cleared set's source while no set was active
|
||||
uint32_t matchedDraws=0;
|
||||
float bestError=INFINITY; // smallest largest-element difference of a position matrix
|
||||
bool bestIndexed=false;
|
||||
std::array<float,12> bestMatrix{};
|
||||
std::array<float,12> expected{};
|
||||
};
|
||||
inline NativeWheelDiagnostics nativeWheelDiagnostics;
|
||||
inline std::vector<const void*> nativeWheelPreviousSources;
|
||||
inline uint32_t nativeWheelClears=0;
|
||||
inline uint32_t nativeWheelReports=0;
|
||||
|
||||
inline void native_wheel_note_outside(const void* source) {
|
||||
for(const void* previous:nativeWheelPreviousSources)
|
||||
if(previous==source) { ++nativeWheelDiagnostics.outsideDraws;return; }
|
||||
}
|
||||
|
||||
inline void native_wheel_note_bound(const NativeWheelArray& replacement,bool indexedMatrix) {
|
||||
auto& diagnostics=nativeWheelDiagnostics;
|
||||
++diagnostics.boundDraws;
|
||||
for(uint32_t slot=0;slot<MaxPnMtx;++slot) {
|
||||
if(!indexedMatrix && slot!=g_gxState.currentPnMtx) continue;
|
||||
const auto* matrix=reinterpret_cast<const float*>(&g_gxState.pnMtx[slot].pos);
|
||||
float error=0;
|
||||
for(int i=0;i<12;++i) error=std::max(error,std::abs(matrix[i]-replacement.modelView[i]));
|
||||
if(error<diagnostics.bestError) {
|
||||
diagnostics.bestError=error;
|
||||
diagnostics.bestIndexed=indexedMatrix;
|
||||
std::memcpy(diagnostics.bestMatrix.data(),matrix,sizeof(float)*12);
|
||||
diagnostics.expected=replacement.modelView;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Called as a set is cleared (GXAurora.cpp): one host log line at about half a
|
||||
// second, five seconds and a minute of sets.
|
||||
void native_wheel_report();
|
||||
inline bool native_wheel_source(const void* source) {
|
||||
for(const auto& replacement:nativeWheelArrays) if(source==replacement.source) return true;
|
||||
return false;
|
||||
}
|
||||
inline NativeWheelArray* native_wheel_array(const AttrArray& array,const uint8_t* vertices,
|
||||
uint32_t vertexBytes,uint32_t vertexStride,uint32_t positionOffset) {
|
||||
const bool indexedMatrix=g_gxState.vtxDesc[GX_VA_PNMTXIDX]==GX_DIRECT;
|
||||
if(!indexedMatrix && g_gxState.currentPnMtx>=MaxPnMtx) return nullptr;
|
||||
for(auto& replacement:nativeWheelArrays) {
|
||||
if(array.data!=replacement.source) continue;
|
||||
if(array.size>replacement.bytes.size()) { ++nativeWheelDiagnostics.oversizeDraws;continue; }
|
||||
native_wheel_note_bound(replacement,indexedMatrix);
|
||||
// A matching asset alone would also animate an opponent. Multi-joint
|
||||
// models need a per-position ownership check, not a blanket exclusion.
|
||||
uint16_t matching=0;
|
||||
for(uint32_t slot=0;slot<MaxPnMtx;++slot) {
|
||||
if(!indexedMatrix && slot!=g_gxState.currentPnMtx) continue;
|
||||
const auto* matrix=reinterpret_cast<const float*>(&g_gxState.pnMtx[slot].pos);
|
||||
bool matches=true;
|
||||
for(int i=0;i<12;++i) {
|
||||
uint32_t bits;std::memcpy(&bits,&matrix[i],4);
|
||||
if((bits&0x7f800000u)==0x7f800000u ||
|
||||
std::abs(matrix[i]-replacement.modelView[i])>(i%4==3?0.1f:0.002f)) { matches=false;break; }
|
||||
}
|
||||
if(matches) matching|=uint16_t(1u<<slot);
|
||||
}
|
||||
if(!matching) continue;
|
||||
if(indexedMatrix && !NativeWheelDrawMatches(
|
||||
{static_cast<const uint8_t*>(array.data),array.size},
|
||||
{replacement.bytes.data(),array.size},array.stride,
|
||||
{vertices,vertexBytes},vertexStride,positionOffset,
|
||||
g_gxState.vtxDesc[GX_VA_POS]==GX_INDEX8?1:2,matching)) continue;
|
||||
++nativeWheelMatches;++nativeWheelDiagnostics.matchedDraws;return &replacement;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,9 @@ struct DrawData {
|
||||
// Same GX state with exact fragment-depth export enabled. Bound only when
|
||||
// this draw is actually reprojected onto the VR virtual screen.
|
||||
gfx::PipelineRef exactScreenDepthPipeline;
|
||||
// Eye depth/stencil format siblings. Shader modules are shared with mono.
|
||||
gfx::PipelineRef stereoPipeline = 0;
|
||||
gfx::PipelineRef stereoScreenPipeline = 0;
|
||||
gfx::Range vertRange;
|
||||
gfx::Range idxRange;
|
||||
gfx::Range uniformRange;
|
||||
@@ -29,7 +32,7 @@ struct DrawData {
|
||||
std::optional<gfx::stereo_replay::SubviewRect> screenRect;
|
||||
};
|
||||
|
||||
constexpr uint32_t GXPipelineConfigVersion = 20;
|
||||
constexpr uint32_t GXPipelineConfigVersion = 21;
|
||||
|
||||
constexpr GXFogType effective_pipeline_fog_type(GXFogType fogType, GXZTexOp zTextureOp, bool zCompLocBeforeTex,
|
||||
GXBlendMode blendMode, GXLogicOp logicOp) noexcept {
|
||||
@@ -41,6 +44,7 @@ constexpr GXFogType effective_pipeline_fog_type(GXFogType fogType, GXZTexOp zTex
|
||||
struct PipelineConfig {
|
||||
uint32_t version = GXPipelineConfigVersion;
|
||||
uint32_t msaaSamples = 1;
|
||||
uint32_t stereoStencil = 0;
|
||||
ShaderConfig shaderConfig;
|
||||
GXCompare depthFunc;
|
||||
GXCullMode cullMode;
|
||||
@@ -61,7 +65,7 @@ inline bool valid_pipeline_config(const PipelineConfig& config) noexcept {
|
||||
};
|
||||
const bool validSamples =
|
||||
config.msaaSamples == 1 || config.msaaSamples == 2 || config.msaaSamples == 4 || config.msaaSamples == 8;
|
||||
return config.version == GXPipelineConfigVersion && validSamples && in_range(config.depthFunc, GX_ALWAYS) &&
|
||||
return config.version == GXPipelineConfigVersion && validSamples && config.stereoStencil <= 1 && in_range(config.depthFunc, GX_ALWAYS) &&
|
||||
in_range(config.cullMode, GX_CULL_ALL) && in_range(config.blendMode, GX_BM_SUBTRACT) &&
|
||||
in_range(config.blendFacSrc, GX_BL_INVDSTALPHA) && in_range(config.blendFacDst, GX_BL_INVDSTALPHA) &&
|
||||
in_range(config.blendOp, GX_LO_SET) && in_range(config.pixelFmt, GX_PF_YUV420);
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
|
||||
#include "fs_helper.hpp"
|
||||
#include "internal.hpp"
|
||||
#include "stereo_overlay.hpp"
|
||||
#include "webgpu/gpu.hpp"
|
||||
#include "window.hpp"
|
||||
|
||||
@@ -28,7 +29,11 @@ static std::string g_imguiLog{};
|
||||
static bool g_useSdlRenderer = false;
|
||||
// Set once ImGui::Render() has produced this frame's draw data. Interpolation encodes up to four
|
||||
// ImGui passes per frame, and every one of them used to rebuild the draw lists from scratch.
|
||||
static bool g_frameDataBuilt = false;
|
||||
static bool g_frameDataBuilt = true;
|
||||
// Host-owned frames (see imgui.hpp). Once the host begins one, aurora never calls new_frame() or
|
||||
// ImGui::Render() itself; the sealed frame carries the host's copy of the draw data instead.
|
||||
static bool g_hostFrames = false;
|
||||
static bool g_hostFrameOpen = false;
|
||||
|
||||
static std::vector<SDL_Texture*> g_sdlTextures;
|
||||
static std::vector<wgpu::Texture> g_wgpuTextures;
|
||||
@@ -179,7 +184,7 @@ void new_frame(const AuroraWindowSize& size) noexcept {
|
||||
|
||||
void render_frame_data() noexcept {
|
||||
ZoneScoped;
|
||||
if (g_frameDataBuilt) {
|
||||
if (g_frameDataBuilt || g_hostFrames) {
|
||||
return;
|
||||
}
|
||||
ImGui::Render();
|
||||
@@ -190,6 +195,11 @@ void render_frame_data() noexcept {
|
||||
|
||||
void render(const wgpu::RenderPassEncoder& pass) noexcept {
|
||||
ZoneScoped;
|
||||
if (g_hostFrames) {
|
||||
// The shared context's draw data belongs to the host's current frame now;
|
||||
// a sealed frame without a host copy has nothing safe to draw.
|
||||
return;
|
||||
}
|
||||
render_frame_data();
|
||||
|
||||
auto* data = ImGui::GetDrawData();
|
||||
@@ -205,6 +215,71 @@ void render(const wgpu::RenderPassEncoder& pass) noexcept {
|
||||
}
|
||||
}
|
||||
|
||||
struct HostFrame {
|
||||
ImDrawData data{};
|
||||
std::vector<ImDrawList*> lists;
|
||||
~HostFrame() {
|
||||
for (ImDrawList* list : lists) {
|
||||
IM_DELETE(list);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
void host_frame_begin(const AuroraWindowSize& size) noexcept {
|
||||
g_hostFrames = true;
|
||||
if (g_hostFrameOpen) {
|
||||
return;
|
||||
}
|
||||
if (!g_frameDataBuilt) {
|
||||
// aurora started this frame itself before the host took over: adopt it.
|
||||
g_hostFrameOpen = true;
|
||||
return;
|
||||
}
|
||||
new_frame(size);
|
||||
g_hostFrameOpen = true;
|
||||
}
|
||||
|
||||
HostFramePtr host_frame_end() noexcept {
|
||||
ZoneScoped;
|
||||
if (!g_hostFrameOpen) {
|
||||
host_frame_begin(window::get_window_size());
|
||||
}
|
||||
ImGui::Render();
|
||||
ImDrawData* source = ImGui::GetDrawData();
|
||||
source->FramebufferScale = ImGui::GetIO().DisplayFramebufferScale;
|
||||
auto frame = std::make_shared<HostFrame>();
|
||||
frame->data = *source;
|
||||
frame->data.CmdLists.clear();
|
||||
frame->lists.reserve(static_cast<size_t>(source->CmdListsCount));
|
||||
for (int i = 0; i < source->CmdListsCount; ++i) {
|
||||
const ImDrawList* src = source->CmdLists[i];
|
||||
ImDrawList* copy = IM_NEW(ImDrawList)(src->_Data);
|
||||
copy->CmdBuffer = src->CmdBuffer;
|
||||
copy->IdxBuffer = src->IdxBuffer;
|
||||
copy->VtxBuffer = src->VtxBuffer;
|
||||
copy->Flags = src->Flags;
|
||||
frame->lists.push_back(copy);
|
||||
frame->data.CmdLists.push_back(copy);
|
||||
}
|
||||
g_hostFrameOpen = false;
|
||||
g_frameDataBuilt = true;
|
||||
return frame;
|
||||
}
|
||||
|
||||
bool host_frames_active() noexcept { return g_hostFrames; }
|
||||
|
||||
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept { return &frame.data; }
|
||||
|
||||
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept {
|
||||
ZoneScoped;
|
||||
if (g_useSdlRenderer || data == nullptr) {
|
||||
return;
|
||||
}
|
||||
pass.PushDebugGroup("Aurora: Dear Imgui");
|
||||
ImGui_ImplWGPU_RenderDrawData(const_cast<ImDrawData*>(data), pass.Get());
|
||||
pass.PopDebugGroup();
|
||||
}
|
||||
|
||||
StereoOverlay latch_stereo_overlay() noexcept {
|
||||
std::lock_guard lock(g_stereoOverlayMutex);
|
||||
return g_stereoOverlay;
|
||||
@@ -274,6 +349,8 @@ ImTextureID aurora_imgui_add_texture(uint32_t width, uint32_t height, const void
|
||||
return aurora::imgui::add_texture(width, height, static_cast<const uint8_t*>(rgba8));
|
||||
}
|
||||
|
||||
void aurora_set_stereo_panel_layer(bool enabled) { aurora::stereo_overlay::set_layer_mode(enabled); }
|
||||
|
||||
void aurora_imgui_set_stereo_overlay(ImDrawData* drawData, float widthFraction) {
|
||||
std::lock_guard lock(aurora::imgui::g_stereoOverlayMutex);
|
||||
aurora::imgui::g_stereoOverlay = {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <aurora/event.h>
|
||||
#include <memory>
|
||||
|
||||
union SDL_Event;
|
||||
struct ImDrawData;
|
||||
@@ -32,4 +33,17 @@ StereoOverlay latch_stereo_overlay() noexcept;
|
||||
// uniform for every pass, so a pass whose display size differs from the desktop's must be submitted
|
||||
// before the next pass is recorded.
|
||||
bool render_draw_data(const wgpu::RenderPassEncoder& pass, ImDrawData* data) noexcept;
|
||||
|
||||
// Host-owned ImGui frames. The host starts each frame on its own thread with host_frame_begin() and
|
||||
// closes it with host_frame_end(), which renders the frame and copies its draw data out of the shared
|
||||
// context. The copy is what the sealed frame replays, so the host may start the next frame while the
|
||||
// worker still encodes this one, and aurora stops starting frames itself once the host has begun one.
|
||||
struct HostFrame;
|
||||
using HostFramePtr = std::shared_ptr<HostFrame>;
|
||||
void host_frame_begin(const AuroraWindowSize& size) noexcept;
|
||||
HostFramePtr host_frame_end() noexcept;
|
||||
bool host_frames_active() noexcept;
|
||||
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept;
|
||||
// Renders a host frame's copied draw data in place of the shared context's.
|
||||
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept;
|
||||
} // namespace aurora::imgui
|
||||
@@ -11,6 +11,7 @@
|
||||
#include "tracy/Tracy.hpp"
|
||||
|
||||
#include <array>
|
||||
#include <atomic>
|
||||
#include <cmath>
|
||||
|
||||
namespace aurora::stereo_overlay {
|
||||
@@ -74,8 +75,12 @@ struct State {
|
||||
std::array<wgpu::BindGroup, AURORA_STEREO_EYE_COUNT> bindGroups;
|
||||
float widthFraction = 0.f;
|
||||
bool visible = false;
|
||||
// Stands in for the panel in a layer while it is not showing.
|
||||
webgpu::TextureWithSampler transparent;
|
||||
bool transparentCleared = false;
|
||||
};
|
||||
State g_state;
|
||||
std::atomic_bool g_layerMode{false};
|
||||
|
||||
bool ensure_pipeline() {
|
||||
auto& state = g_state;
|
||||
@@ -239,6 +244,7 @@ void composite(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& tar
|
||||
.label = eyeIndex == 0 ? "Headset panel left eye" : "Headset panel right eye",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
|
||||
};
|
||||
const auto pass = encoder.BeginRenderPass(&descriptor);
|
||||
pass.SetPipeline(state.pipeline);
|
||||
@@ -294,6 +300,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
|
||||
.label = "Headset panel ImGui pass",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
.timestampWrites = gfx::gpu_timing_pass(gfx::GpuTimingCategory::Panel),
|
||||
};
|
||||
bool drawn = false;
|
||||
{
|
||||
@@ -312,7 +319,7 @@ wgpu::CommandBuffer prepare(ImDrawData* drawData, float widthFraction) noexcept
|
||||
void composite_immersive(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& eye,
|
||||
const Mat4x4<float>& eyeFrustum, const Mat3x4<float>& viewFromCenter,
|
||||
uint32_t eyeIndex) noexcept {
|
||||
if (!g_state.visible) {
|
||||
if (!g_state.visible || layer_mode()) {
|
||||
return;
|
||||
}
|
||||
float screenWidth = 0.f;
|
||||
@@ -329,7 +336,7 @@ void composite_immersive(const wgpu::CommandEncoder& encoder, const wgpu::Textur
|
||||
|
||||
void composite_flat(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& eye, const wgpu::Extent3D& size,
|
||||
uint32_t eyeIndex) noexcept {
|
||||
if (!g_state.visible || size.width == 0 || size.height == 0) {
|
||||
if (!g_state.visible || layer_mode() || size.width == 0 || size.height == 0) {
|
||||
return;
|
||||
}
|
||||
const float imageAspect = static_cast<float>(size.width) / static_cast<float>(size.height);
|
||||
@@ -338,6 +345,52 @@ void composite_flat(const wgpu::CommandEncoder& encoder, const wgpu::TextureView
|
||||
eyeIndex);
|
||||
}
|
||||
|
||||
void set_layer_mode(bool enabled) noexcept { g_layerMode.store(enabled, std::memory_order_release); }
|
||||
|
||||
bool layer_mode() noexcept { return g_layerMode.load(std::memory_order_acquire); }
|
||||
|
||||
bool layer_source(const wgpu::CommandEncoder& encoder, uint32_t width, uint32_t height, stereo::EyeImage& out) noexcept {
|
||||
auto& state = g_state;
|
||||
const auto format = webgpu::g_graphicsConfig.surfaceConfiguration.format;
|
||||
if (width == 0 || height == 0) {
|
||||
return false;
|
||||
}
|
||||
if (state.visible && state.panel.texture && state.panel.size.width == width && state.panel.size.height == height &&
|
||||
state.panel.format == format) {
|
||||
out = {.texture = &state.panel.texture, .view = &state.panel.view, .size = state.panel.size, .format = format};
|
||||
return true;
|
||||
}
|
||||
if (!state.transparent.texture || state.transparent.size.width != width ||
|
||||
state.transparent.size.height != height || state.transparent.format != format) {
|
||||
state.transparent = webgpu::create_render_texture(width, height, false);
|
||||
state.transparentCleared = false;
|
||||
if (state.transparent.size.width != width || state.transparent.size.height != height) {
|
||||
state.transparent = {};
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!state.transparentCleared) {
|
||||
const std::array attachments{
|
||||
wgpu::RenderPassColorAttachment{
|
||||
.view = state.transparent.view,
|
||||
.loadOp = wgpu::LoadOp::Clear,
|
||||
.storeOp = wgpu::StoreOp::Store,
|
||||
.clearValue = {.r = 0.0, .g = 0.0, .b = 0.0, .a = 0.0},
|
||||
},
|
||||
};
|
||||
const wgpu::RenderPassDescriptor descriptor{
|
||||
.label = "Headset panel layer clear",
|
||||
.colorAttachmentCount = attachments.size(),
|
||||
.colorAttachments = attachments.data(),
|
||||
};
|
||||
encoder.BeginRenderPass(&descriptor).End();
|
||||
state.transparentCleared = true;
|
||||
}
|
||||
out = {.texture = &state.transparent.texture, .view = &state.transparent.view, .size = state.transparent.size,
|
||||
.format = format};
|
||||
return true;
|
||||
}
|
||||
|
||||
void shutdown() noexcept { g_state = {}; }
|
||||
|
||||
} // namespace aurora::stereo_overlay
|
||||
@@ -3,6 +3,8 @@
|
||||
#include <aurora/math.hpp>
|
||||
#include <webgpu/webgpu_cpp.h>
|
||||
|
||||
#include "stereo.hpp"
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
struct ImDrawData;
|
||||
@@ -28,6 +30,20 @@ void composite_immersive(const wgpu::CommandEncoder& encoder, const wgpu::Textur
|
||||
void composite_flat(const wgpu::CommandEncoder& encoder, const wgpu::TextureView& eye, const wgpu::Extent3D& size,
|
||||
uint32_t eyeIndex) noexcept;
|
||||
|
||||
// Layer mode: the OpenXR backend shows the panel as its own compositor quad
|
||||
// layer, sharp at any eye resolution, so it is no longer drawn into the eyes
|
||||
// (both composite functions do nothing). Any thread; read by the frame worker.
|
||||
void set_layer_mode(bool enabled) noexcept;
|
||||
bool layer_mode() noexcept;
|
||||
|
||||
// Frame worker, inside a stereo sink: the image to copy into a panel layer of
|
||||
// width x height in the eyes' format. That is the panel while it is showing at
|
||||
// exactly that size, and otherwise a transparent image of that size, so a layer
|
||||
// asked for before the panel's first frame (or after it closed) shows nothing.
|
||||
// A transparent image is cleared once, by a pass recorded into `encoder`.
|
||||
// False only when no image can be made.
|
||||
bool layer_source(const wgpu::CommandEncoder& encoder, uint32_t width, uint32_t height, stereo::EyeImage& out) noexcept;
|
||||
|
||||
void shutdown() noexcept;
|
||||
|
||||
} // namespace aurora::stereo_overlay
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
#include "../internal.hpp"
|
||||
#include "../stereo.hpp"
|
||||
#include "../stereo_overlay.hpp"
|
||||
#include "gpu.hpp"
|
||||
|
||||
#if defined(_WIN32) && defined(WEBGPU_DAWN) && defined(DAWN_ENABLE_BACKEND_D3D12)
|
||||
@@ -126,6 +127,11 @@ struct SharedFenceDxgiHandleWire {
|
||||
void* handle = nullptr;
|
||||
};
|
||||
|
||||
// The eyes, then the settings panel's layer image in a slot of its own so the
|
||||
// eye intermediates are never resized for it.
|
||||
constexpr uint32_t kPanelIndex = AURORA_D3D12_STEREO_MAX_TARGETS;
|
||||
constexpr uint32_t kMaxImages = AURORA_D3D12_STEREO_MAX_TARGETS + 1;
|
||||
|
||||
struct IntermediateEye {
|
||||
ComPtr<ID3D12Resource> resource;
|
||||
wgpu::SharedTextureMemory memory;
|
||||
@@ -153,8 +159,8 @@ struct InFlightCommand {
|
||||
// both sides of every copy alive until this submission's fence completes;
|
||||
// an eye-size change may otherwise replace the bridge intermediate while
|
||||
// the GPU is still reading it.
|
||||
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> sources;
|
||||
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> destinations;
|
||||
std::array<ComPtr<ID3D12Resource>, kMaxImages> sources;
|
||||
std::array<ComPtr<ID3D12Resource>, kMaxImages> destinations;
|
||||
};
|
||||
|
||||
class StereoBridge final {
|
||||
@@ -209,38 +215,51 @@ public:
|
||||
return WaitForGpuLocked();
|
||||
}
|
||||
|
||||
bool SetTargets(uint64_t token, const AuroraD3D12StereoTarget* targets,
|
||||
uint32_t targetCount) noexcept {
|
||||
bool SetTargets(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t targetCount,
|
||||
const AuroraD3D12StereoTarget* panel) noexcept {
|
||||
if (token == 0 || targets == nullptr || targetCount == 0 ||
|
||||
targetCount > AURORA_D3D12_STEREO_MAX_TARGETS) {
|
||||
return false;
|
||||
}
|
||||
const auto valid = [](const AuroraD3D12StereoTarget& target) {
|
||||
if (target.resource == nullptr || target.width == 0 || target.height == 0 ||
|
||||
target.dxgiFormat == DXGI_FORMAT_UNKNOWN) {
|
||||
return false;
|
||||
}
|
||||
const D3D12_RESOURCE_DESC desc = static_cast<ID3D12Resource*>(target.resource)->GetDesc();
|
||||
return desc.Dimension == D3D12_RESOURCE_DIMENSION_TEXTURE2D && desc.Width >= target.width &&
|
||||
desc.Height >= target.height && desc.DepthOrArraySize == 1 && desc.MipLevels == 1 &&
|
||||
desc.SampleDesc.Count == 1 &&
|
||||
same_copy_family(desc.Format, static_cast<DXGI_FORMAT>(target.dxgiFormat));
|
||||
};
|
||||
std::lock_guard lock(m_mutex);
|
||||
if (m_framePending || m_encoded) {
|
||||
return false;
|
||||
}
|
||||
for (uint32_t eye = 0; eye < targetCount; ++eye) {
|
||||
if (targets[eye].resource == nullptr || targets[eye].width == 0 ||
|
||||
targets[eye].height == 0 || targets[eye].dxgiFormat == DXGI_FORMAT_UNKNOWN) {
|
||||
if (!valid(targets[eye])) {
|
||||
return false;
|
||||
}
|
||||
auto* resource = static_cast<ID3D12Resource*>(targets[eye].resource);
|
||||
const D3D12_RESOURCE_DESC desc = resource->GetDesc();
|
||||
if (desc.Dimension != D3D12_RESOURCE_DIMENSION_TEXTURE2D ||
|
||||
desc.Width < targets[eye].width || desc.Height < targets[eye].height ||
|
||||
desc.DepthOrArraySize != 1 || desc.MipLevels != 1 || desc.SampleDesc.Count != 1 ||
|
||||
!same_copy_family(desc.Format, static_cast<DXGI_FORMAT>(targets[eye].dxgiFormat))) {
|
||||
return false;
|
||||
}
|
||||
m_targets[eye] = {
|
||||
.resource = resource,
|
||||
.width = targets[eye].width,
|
||||
.height = targets[eye].height,
|
||||
.format = static_cast<DXGI_FORMAT>(targets[eye].dxgiFormat),
|
||||
};
|
||||
}
|
||||
for (uint32_t eye = targetCount; eye < m_targets.size(); ++eye) {
|
||||
m_targets[eye] = {};
|
||||
if (panel != nullptr && !valid(*panel)) {
|
||||
return false;
|
||||
}
|
||||
m_targets = {};
|
||||
m_imageCount = 0;
|
||||
const auto add = [&](uint32_t index, const AuroraD3D12StereoTarget& target) {
|
||||
m_targets[index] = {
|
||||
.resource = static_cast<ID3D12Resource*>(target.resource),
|
||||
.width = target.width,
|
||||
.height = target.height,
|
||||
.format = static_cast<DXGI_FORMAT>(target.dxgiFormat),
|
||||
};
|
||||
m_images[m_imageCount++] = index;
|
||||
};
|
||||
for (uint32_t eye = 0; eye < targetCount; ++eye) {
|
||||
add(eye, targets[eye]);
|
||||
}
|
||||
if (panel != nullptr) {
|
||||
add(kPanelIndex, *panel);
|
||||
}
|
||||
m_frameToken = token;
|
||||
m_targetCount = targetCount;
|
||||
@@ -307,7 +326,7 @@ private:
|
||||
if (source.texture == nullptr || sourceFormat == DXGI_FORMAT_UNKNOWN ||
|
||||
source.size.width != m_targets[eye].width || source.size.height != m_targets[eye].height ||
|
||||
!same_copy_family(sourceFormat, m_targets[eye].format)) {
|
||||
Log.error("Stereo eye {} does not match its OpenXR D3D12 target", eye);
|
||||
Log.error("Stereo image {} does not match its OpenXR D3D12 target", eye);
|
||||
return false;
|
||||
}
|
||||
if (intermediate.texture && intermediate.width == source.size.width &&
|
||||
@@ -350,7 +369,9 @@ private:
|
||||
wire.resource = intermediate.resource;
|
||||
const wgpu::SharedTextureMemoryDescriptor memoryDescriptor{
|
||||
.nextInChain = &wire.chain,
|
||||
.label = eye == 0 ? "OpenXR left eye intermediate" : "OpenXR right eye intermediate",
|
||||
.label = eye == 0 ? "OpenXR left eye intermediate"
|
||||
: eye == 1 ? "OpenXR right eye intermediate"
|
||||
: "OpenXR panel intermediate",
|
||||
};
|
||||
intermediate.memory = webgpu::g_device.ImportSharedTextureMemory(&memoryDescriptor);
|
||||
if (!intermediate.memory) {
|
||||
@@ -368,7 +389,9 @@ private:
|
||||
return false;
|
||||
}
|
||||
const wgpu::TextureDescriptor textureDescriptor{
|
||||
.label = eye == 0 ? "OpenXR left eye shared texture" : "OpenXR right eye shared texture",
|
||||
.label = eye == 0 ? "OpenXR left eye shared texture"
|
||||
: eye == 1 ? "OpenXR right eye shared texture"
|
||||
: "OpenXR panel shared texture",
|
||||
.usage = wgpu::TextureUsage::CopyDst,
|
||||
.dimension = wgpu::TextureDimension::e2D,
|
||||
.size = {source.size.width, source.size.height, 1},
|
||||
@@ -391,12 +414,22 @@ private:
|
||||
|
||||
bool EncodeLocked(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) noexcept {
|
||||
CollectCompletedCommandsLocked();
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
if (!EnsureIntermediate(eye, frame.eyes[eye])) {
|
||||
std::array<stereo::EyeImage, kMaxImages> sources{};
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
if (eye == kPanelIndex) {
|
||||
if (!stereo_overlay::layer_source(encoder, m_targets[eye].width, m_targets[eye].height, sources[eye])) {
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
sources[eye] = frame.eyes[eye];
|
||||
}
|
||||
if (!EnsureIntermediate(eye, sources[eye])) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
auto& intermediate = m_intermediates[eye];
|
||||
const std::array fences{m_webgpuFence};
|
||||
const std::array values{m_lastExternalFenceValue};
|
||||
@@ -410,7 +443,8 @@ private:
|
||||
}
|
||||
if (intermediate.memory.BeginAccess(intermediate.texture, &begin) != wgpu::Status::Success) {
|
||||
Log.error("Dawn BeginAccess failed for stereo eye {}", eye);
|
||||
for (uint32_t begunEye = 0; begunEye < eye; ++begunEye) {
|
||||
for (uint32_t m = 0; m < n; ++m) {
|
||||
const uint32_t begunEye = m_images[m];
|
||||
wgpu::SharedTextureMemoryEndAccessState end{};
|
||||
m_intermediates[begunEye].memory.EndAccess(m_intermediates[begunEye].texture, &end);
|
||||
m_intermediates[begunEye].initialized = end.initialized;
|
||||
@@ -424,10 +458,11 @@ private:
|
||||
// to one. If a later BeginAccess fails, the rollback above can therefore
|
||||
// end the earlier accesses without leaving an unsubmitted copy that uses
|
||||
// a texture after its access interval.
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
const auto& intermediate = m_intermediates[eye];
|
||||
const wgpu::TexelCopyTextureInfo source{
|
||||
.texture = *frame.eyes[eye].texture,
|
||||
.texture = *sources[eye].texture,
|
||||
.mipLevel = 0,
|
||||
.origin = {},
|
||||
.aspect = wgpu::TextureAspect::All,
|
||||
@@ -446,7 +481,8 @@ private:
|
||||
|
||||
bool EndAccessLocked() noexcept {
|
||||
bool success = true;
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
auto& intermediate = m_intermediates[eye];
|
||||
if (!intermediate.accessBegun) {
|
||||
success = false;
|
||||
@@ -467,8 +503,8 @@ private:
|
||||
bool EnqueueNativeCopyLocked() noexcept {
|
||||
ComPtr<ID3D12CommandAllocator> allocator;
|
||||
ComPtr<ID3D12GraphicsCommandList> list;
|
||||
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> sources;
|
||||
std::array<ComPtr<ID3D12Resource>, AURORA_D3D12_STEREO_MAX_TARGETS> destinations;
|
||||
std::array<ComPtr<ID3D12Resource>, kMaxImages> sources;
|
||||
std::array<ComPtr<ID3D12Resource>, kMaxImages> destinations;
|
||||
if (FAILED(m_device->CreateCommandAllocator(D3D12_COMMAND_LIST_TYPE_DIRECT,
|
||||
IID_PPV_ARGS(&allocator))) ||
|
||||
FAILED(m_device->CreateCommandList(0, D3D12_COMMAND_LIST_TYPE_DIRECT, allocator.Get(),
|
||||
@@ -477,7 +513,8 @@ private:
|
||||
return false;
|
||||
}
|
||||
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
const auto& source = m_intermediates[eye];
|
||||
const auto& destination = m_targets[eye];
|
||||
sources[eye] = source.resource;
|
||||
@@ -564,6 +601,7 @@ private:
|
||||
}
|
||||
m_frameToken = 0;
|
||||
m_targetCount = 0;
|
||||
m_imageCount = 0;
|
||||
m_framePending = false;
|
||||
m_encoded = false;
|
||||
}
|
||||
@@ -625,8 +663,11 @@ private:
|
||||
ComPtr<ID3D12CommandQueue> m_queue;
|
||||
ComPtr<ID3D12Fence> m_fence;
|
||||
wgpu::SharedFence m_webgpuFence;
|
||||
std::array<IntermediateEye, AURORA_D3D12_STEREO_MAX_TARGETS> m_intermediates{};
|
||||
std::array<PendingTarget, AURORA_D3D12_STEREO_MAX_TARGETS> m_targets{};
|
||||
std::array<IntermediateEye, kMaxImages> m_intermediates{};
|
||||
std::array<PendingTarget, kMaxImages> m_targets{};
|
||||
// The slots of m_targets this frame copies into, eyes first.
|
||||
std::array<uint32_t, kMaxImages> m_images{};
|
||||
uint32_t m_imageCount = 0;
|
||||
std::vector<InFlightCommand> m_commands;
|
||||
AuroraD3D12StereoSubmittedCallback m_callback = nullptr;
|
||||
void* m_userdata = nullptr;
|
||||
@@ -701,7 +742,13 @@ bool aurora_d3d12_set_stereo_targets(uint64_t frameToken,
|
||||
const AuroraD3D12StereoTarget* targets,
|
||||
uint32_t targetCount) {
|
||||
using namespace aurora::d3d12_interop;
|
||||
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount);
|
||||
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, nullptr);
|
||||
}
|
||||
|
||||
bool aurora_d3d12_set_stereo_targets_with_panel(uint64_t frameToken, const AuroraD3D12StereoTarget* targets,
|
||||
uint32_t targetCount, const AuroraD3D12StereoTarget* panel) {
|
||||
using namespace aurora::d3d12_interop;
|
||||
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, panel);
|
||||
}
|
||||
|
||||
bool aurora_d3d12_cancel_stereo_targets(uint64_t frameToken) {
|
||||
@@ -744,6 +791,11 @@ bool aurora_d3d12_set_stereo_targets(uint64_t, const AuroraD3D12StereoTarget*, u
|
||||
return false;
|
||||
}
|
||||
|
||||
bool aurora_d3d12_set_stereo_targets_with_panel(uint64_t, const AuroraD3D12StereoTarget*, uint32_t,
|
||||
const AuroraD3D12StereoTarget*) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool aurora_d3d12_cancel_stereo_targets(uint64_t) { return false; }
|
||||
|
||||
bool aurora_d3d12_disable_stereo_bridge() { return true; }
|
||||
|
||||
@@ -80,6 +80,7 @@ wgpu::Instance g_instance;
|
||||
static wgpu::AdapterInfo g_adapterInfo;
|
||||
static wgpu::SurfaceCapabilities g_surfaceCapabilities;
|
||||
bool g_bcTexturesSupported;
|
||||
bool g_timestampQueriesSupported = false;
|
||||
// Written by Dawn's device-loss callback and consumed at ordered frame boundaries. Keep the
|
||||
// callback free of logging, allocation, teardown and renderer state mutation.
|
||||
static std::atomic_bool g_deviceLost{false};
|
||||
@@ -702,6 +703,12 @@ bool initialize(AuroraBackend auroraBackend) {
|
||||
g_bcTexturesSupported = true;
|
||||
requiredFeatures.push_back(feature);
|
||||
}
|
||||
// Per-pass GPU timing for the frame-rate log (gfx::gpu_timing_*). Requesting the feature
|
||||
// costs nothing until a pass carries timestamp writes.
|
||||
if (feature == wgpu::FeatureName::TimestampQuery) {
|
||||
g_timestampQueriesSupported = true;
|
||||
requiredFeatures.push_back(feature);
|
||||
}
|
||||
// The presenter calls device and queue methods while the frame worker encodes, which Dawn only
|
||||
// supports with this feature; without it the two race inside the device's dynamic uploader.
|
||||
if (feature == wgpu::FeatureName::ImplicitDeviceSynchronization) {
|
||||
@@ -769,10 +776,15 @@ bool initialize(AuroraBackend auroraBackend) {
|
||||
if (g_backendType == wgpu::BackendType::Vulkan) {
|
||||
enableToggles.push_back("vulkan_monolithic_pipeline_cache");
|
||||
}
|
||||
// Dawn quantizes timestamp queries to 100 us for web privacy; the per-pass GPU timing wants
|
||||
// the raw values.
|
||||
const std::array<const char*, 1> disableToggles{"timestamp_quantization"};
|
||||
const wgpu::DawnTogglesDescriptor togglesDescriptor({
|
||||
.nextInChain = &cacheDescriptor,
|
||||
.enabledToggleCount = enableToggles.size(),
|
||||
.enabledToggles = enableToggles.data(),
|
||||
.disabledToggleCount = g_timestampQueriesSupported ? disableToggles.size() : 0,
|
||||
.disabledToggles = disableToggles.data(),
|
||||
});
|
||||
#endif
|
||||
wgpu::DeviceDescriptor deviceDescriptor;
|
||||
|
||||
@@ -58,6 +58,8 @@ extern wgpu::RenderPipeline g_CopyPipeline;
|
||||
extern wgpu::BindGroup g_CopyBindGroup;
|
||||
extern wgpu::Instance g_instance;
|
||||
extern bool g_bcTexturesSupported;
|
||||
// The device was created with TimestampQuery, so passes may carry timestamp writes (gfx::gpu_timing_*).
|
||||
extern bool g_timestampQueriesSupported;
|
||||
|
||||
bool initialize(AuroraBackend backend);
|
||||
void shutdown();
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
#include "../internal.hpp"
|
||||
#include "../stereo.hpp"
|
||||
#include "../stereo_overlay.hpp"
|
||||
#include "gpu.hpp"
|
||||
|
||||
#if defined(__ANDROID__) && defined(WEBGPU_DAWN)
|
||||
@@ -105,6 +106,11 @@ struct Import {
|
||||
bool accessBegun = false;
|
||||
};
|
||||
|
||||
// The eyes, then the settings panel's layer image in a slot of its own.
|
||||
constexpr uint32_t kPanelIndex = AURORA_VULKAN_STEREO_MAX_TARGETS;
|
||||
constexpr uint32_t kMaxImages = AURORA_VULKAN_STEREO_MAX_RELEASES;
|
||||
using Releases = std::array<AuroraVulkanStereoRelease, kMaxImages>;
|
||||
|
||||
struct PendingTarget {
|
||||
AHardwareBuffer* buffer = nullptr;
|
||||
uint32_t width = 0;
|
||||
@@ -146,8 +152,8 @@ public:
|
||||
return true;
|
||||
}
|
||||
|
||||
bool SetTargets(uint64_t token, const AuroraVulkanStereoTarget* targets,
|
||||
uint32_t targetCount) noexcept {
|
||||
bool SetTargets(uint64_t token, const AuroraVulkanStereoTarget* targets, uint32_t targetCount,
|
||||
const AuroraVulkanStereoTarget* panel) noexcept {
|
||||
if (token == 0 || targets == nullptr || targetCount == 0 ||
|
||||
targetCount > AURORA_VULKAN_STEREO_MAX_TARGETS) {
|
||||
return false;
|
||||
@@ -157,25 +163,36 @@ public:
|
||||
return false;
|
||||
}
|
||||
const int64_t auroraFormat = to_vk_format(m_auroraFormat);
|
||||
const auto valid = [&](const AuroraVulkanStereoTarget& target) {
|
||||
return target.buffer != nullptr && target.width != 0 && target.height != 0 &&
|
||||
same_copy_family(target.vkFormat, auroraFormat);
|
||||
};
|
||||
for (uint32_t eye = 0; eye < targetCount; ++eye) {
|
||||
const auto& target = targets[eye];
|
||||
if (target.buffer == nullptr || target.width == 0 || target.height == 0 ||
|
||||
!same_copy_family(target.vkFormat, auroraFormat)) {
|
||||
if (!valid(targets[eye])) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
for (uint32_t eye = 0; eye < targetCount; ++eye) {
|
||||
m_targets[eye] = {
|
||||
.buffer = targets[eye].buffer,
|
||||
.width = targets[eye].width,
|
||||
.height = targets[eye].height,
|
||||
.vkFormat = targets[eye].vkFormat,
|
||||
.acquireFenceFd = targets[eye].acquireFenceFd,
|
||||
.acquireImageLayout = targets[eye].acquireImageLayout,
|
||||
};
|
||||
if (panel != nullptr && !valid(*panel)) {
|
||||
return false;
|
||||
}
|
||||
for (uint32_t eye = targetCount; eye < m_targets.size(); ++eye) {
|
||||
m_targets[eye] = {};
|
||||
m_targets = {};
|
||||
m_imageCount = 0;
|
||||
const auto add = [&](uint32_t index, const AuroraVulkanStereoTarget& target) {
|
||||
m_targets[index] = {
|
||||
.buffer = target.buffer,
|
||||
.width = target.width,
|
||||
.height = target.height,
|
||||
.vkFormat = target.vkFormat,
|
||||
.acquireFenceFd = target.acquireFenceFd,
|
||||
.acquireImageLayout = target.acquireImageLayout,
|
||||
};
|
||||
m_images[m_imageCount++] = index;
|
||||
};
|
||||
for (uint32_t eye = 0; eye < targetCount; ++eye) {
|
||||
add(eye, targets[eye]);
|
||||
}
|
||||
if (panel != nullptr) {
|
||||
add(kPanelIndex, *panel);
|
||||
}
|
||||
m_frameToken = token;
|
||||
m_targetCount = targetCount;
|
||||
@@ -202,7 +219,7 @@ public:
|
||||
if (!m_framePending || !m_encoded || frame.frameToken != m_frameToken) {
|
||||
return;
|
||||
}
|
||||
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS> releases{};
|
||||
Releases releases{};
|
||||
const bool success = EndAccessLocked(releases);
|
||||
NotifyLocked(frame.frameToken, success, true, releases);
|
||||
ClearFrameLocked();
|
||||
@@ -215,7 +232,7 @@ public:
|
||||
}
|
||||
const uint64_t token = m_frameToken;
|
||||
const bool encoded = m_encoded;
|
||||
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS> releases{};
|
||||
Releases releases{};
|
||||
if (encoded) {
|
||||
EndAccessLocked(releases);
|
||||
for (auto& release : releases) {
|
||||
@@ -242,7 +259,7 @@ private:
|
||||
const auto& target = m_targets[eye];
|
||||
if (source.texture == nullptr || source.format != m_auroraFormat ||
|
||||
source.size.width != target.width || source.size.height != target.height) {
|
||||
Log.error("Stereo eye {} does not match its OpenXR Vulkan target ({}x{} vs {}x{})", eye,
|
||||
Log.error("Stereo image {} does not match its OpenXR Vulkan target ({}x{} vs {}x{})", eye,
|
||||
source.size.width, source.size.height, target.width, target.height);
|
||||
return nullptr;
|
||||
}
|
||||
@@ -264,7 +281,9 @@ private:
|
||||
ahb.handle = target.buffer;
|
||||
const wgpu::SharedTextureMemoryDescriptor memoryDescriptor{
|
||||
.nextInChain = &ahb,
|
||||
.label = eye == 0 ? "OpenXR left eye AHardwareBuffer" : "OpenXR right eye AHardwareBuffer",
|
||||
.label = eye == 0 ? "OpenXR left eye AHardwareBuffer"
|
||||
: eye == 1 ? "OpenXR right eye AHardwareBuffer"
|
||||
: "OpenXR panel AHardwareBuffer",
|
||||
};
|
||||
import.memory = webgpu::g_device.ImportSharedTextureMemory(&memoryDescriptor);
|
||||
if (!import.memory) {
|
||||
@@ -291,7 +310,9 @@ private:
|
||||
return nullptr;
|
||||
}
|
||||
const wgpu::TextureDescriptor textureDescriptor{
|
||||
.label = eye == 0 ? "OpenXR left eye shared texture" : "OpenXR right eye shared texture",
|
||||
.label = eye == 0 ? "OpenXR left eye shared texture"
|
||||
: eye == 1 ? "OpenXR right eye shared texture"
|
||||
: "OpenXR panel shared texture",
|
||||
.usage = wgpu::TextureUsage::CopyDst,
|
||||
.dimension = wgpu::TextureDimension::e2D,
|
||||
.size = {target.width, target.height, 1},
|
||||
@@ -313,19 +334,29 @@ private:
|
||||
}
|
||||
|
||||
bool EncodeLocked(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) noexcept {
|
||||
std::array<Import*, AURORA_VULKAN_STEREO_MAX_TARGETS> imports{};
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
imports[eye] = EnsureImport(eye, frame.eyes[eye]);
|
||||
std::array<stereo::EyeImage, kMaxImages> sources{};
|
||||
std::array<Import*, kMaxImages> imports{};
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
if (eye == kPanelIndex) {
|
||||
if (!stereo_overlay::layer_source(encoder, m_targets[eye].width, m_targets[eye].height, sources[eye])) {
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
sources[eye] = frame.eyes[eye];
|
||||
}
|
||||
imports[eye] = EnsureImport(eye, sources[eye]);
|
||||
if (imports[eye] == nullptr) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
auto& target = m_targets[eye];
|
||||
auto& import = *imports[eye];
|
||||
if (import.accessBegun) {
|
||||
Log.error("AHardwareBuffer for eye {} is still under a previous access", eye);
|
||||
RollbackAccesses(imports, eye);
|
||||
RollbackAccesses(imports, n);
|
||||
return false;
|
||||
}
|
||||
// The OpenXR side's release barrier leaves the image in acquireImageLayout;
|
||||
@@ -350,7 +381,7 @@ private:
|
||||
close_fd(target.acquireFenceFd);
|
||||
if (!acquireFence) {
|
||||
Log.error("Dawn could not import the OpenXR copy-out fence for eye {}", eye);
|
||||
RollbackAccesses(imports, eye);
|
||||
RollbackAccesses(imports, n);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -370,15 +401,16 @@ private:
|
||||
}
|
||||
if (import.memory.BeginAccess(import.texture, &begin) != wgpu::Status::Success) {
|
||||
Log.error("Dawn BeginAccess failed for stereo eye {}", eye);
|
||||
RollbackAccesses(imports, eye);
|
||||
RollbackAccesses(imports, n);
|
||||
return false;
|
||||
}
|
||||
import.accessBegun = true;
|
||||
}
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
const auto& import = *imports[eye];
|
||||
const wgpu::TexelCopyTextureInfo source{
|
||||
.texture = *frame.eyes[eye].texture,
|
||||
.texture = *sources[eye].texture,
|
||||
.mipLevel = 0,
|
||||
.origin = {},
|
||||
.aspect = wgpu::TextureAspect::All,
|
||||
@@ -396,9 +428,10 @@ private:
|
||||
return true;
|
||||
}
|
||||
|
||||
void RollbackAccesses(const std::array<Import*, AURORA_VULKAN_STEREO_MAX_TARGETS>& imports,
|
||||
uint32_t count) noexcept {
|
||||
for (uint32_t eye = 0; eye < count; ++eye) {
|
||||
// Ends the accesses begun for the first `count` images of this frame.
|
||||
void RollbackAccesses(const std::array<Import*, kMaxImages>& imports, uint32_t count) noexcept {
|
||||
for (uint32_t n = 0; n < count; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
if (imports[eye] != nullptr && imports[eye]->accessBegun) {
|
||||
wgpu::SharedTextureMemoryEndAccessState end{};
|
||||
imports[eye]->memory.EndAccess(imports[eye]->texture, &end);
|
||||
@@ -411,13 +444,15 @@ private:
|
||||
}
|
||||
}
|
||||
|
||||
bool EndAccessLocked(
|
||||
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS>& releases) noexcept {
|
||||
// Fills one release per image of this frame, in order: the eyes, then the panel.
|
||||
bool EndAccessLocked(Releases& releases) noexcept {
|
||||
bool success = true;
|
||||
for (auto& release : releases) {
|
||||
release = {.releaseFenceFd = -1, .releasedImageLayout = VK_IMAGE_LAYOUT_UNDEFINED};
|
||||
}
|
||||
for (uint32_t eye = 0; eye < m_targetCount; ++eye) {
|
||||
for (uint32_t n = 0; n < m_imageCount; ++n) {
|
||||
const uint32_t eye = m_images[n];
|
||||
auto& release = releases[n];
|
||||
Import* import = m_encodedImports[eye];
|
||||
if (import == nullptr || !import->accessBegun) {
|
||||
success = false;
|
||||
@@ -431,7 +466,7 @@ private:
|
||||
success = false;
|
||||
} else {
|
||||
import->initialized = end.initialized;
|
||||
releases[eye].releasedImageLayout = layout.newLayout;
|
||||
release.releasedImageLayout = layout.newLayout;
|
||||
for (size_t i = 0; i < end.fenceCount; ++i) {
|
||||
wgpu::SharedFenceSyncFDExportInfo syncFd{};
|
||||
wgpu::SharedFenceExportInfo info{};
|
||||
@@ -441,16 +476,16 @@ private:
|
||||
// The fence keeps its descriptor; hand the caller an independent one.
|
||||
const int duplicate = ::dup(syncFd.handle);
|
||||
if (duplicate >= 0) {
|
||||
if (releases[eye].releaseFenceFd >= 0) {
|
||||
if (release.releaseFenceFd >= 0) {
|
||||
// Dawn normally returns exactly one fence per access. Both must
|
||||
// be honoured and one descriptor cannot express two fences, so
|
||||
// the earlier one is retired on the CPU before handing over the
|
||||
// latest.
|
||||
Log.warn("Dawn returned several release fences for eye {}; merging on the CPU", eye);
|
||||
wait_sync_fd(releases[eye].releaseFenceFd);
|
||||
close_fd(releases[eye].releaseFenceFd);
|
||||
wait_sync_fd(release.releaseFenceFd);
|
||||
close_fd(release.releaseFenceFd);
|
||||
}
|
||||
releases[eye].releaseFenceFd = duplicate;
|
||||
release.releaseFenceFd = duplicate;
|
||||
}
|
||||
} else {
|
||||
Log.error("Dawn returned a non-sync-fd fence for eye {}", eye);
|
||||
@@ -482,12 +517,13 @@ private:
|
||||
m_encodedImports = {};
|
||||
m_frameToken = 0;
|
||||
m_targetCount = 0;
|
||||
m_imageCount = 0;
|
||||
m_framePending = false;
|
||||
m_encoded = false;
|
||||
}
|
||||
|
||||
void PublishAndClearFrameLocked(uint64_t token, bool success, bool gpuWorkQueued) noexcept {
|
||||
std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS> releases{};
|
||||
Releases releases{};
|
||||
for (auto& release : releases) {
|
||||
release = {.releaseFenceFd = -1, .releasedImageLayout = VK_IMAGE_LAYOUT_UNDEFINED};
|
||||
}
|
||||
@@ -496,10 +532,9 @@ private:
|
||||
}
|
||||
|
||||
void NotifyLocked(uint64_t token, bool success, bool gpuWorkQueued,
|
||||
const std::array<AuroraVulkanStereoRelease, AURORA_VULKAN_STEREO_MAX_TARGETS>&
|
||||
releases) noexcept {
|
||||
const Releases& releases) noexcept {
|
||||
if (m_callback != nullptr) {
|
||||
m_callback(token, success, gpuWorkQueued, releases.data(), m_targetCount, m_userdata);
|
||||
m_callback(token, success, gpuWorkQueued, releases.data(), m_imageCount, m_userdata);
|
||||
} else {
|
||||
for (auto release : releases) {
|
||||
close_fd(release.releaseFenceFd);
|
||||
@@ -509,8 +544,11 @@ private:
|
||||
|
||||
std::mutex m_mutex;
|
||||
std::unordered_map<AHardwareBuffer*, Import> m_imports;
|
||||
std::array<PendingTarget, AURORA_VULKAN_STEREO_MAX_TARGETS> m_targets{};
|
||||
std::array<Import*, AURORA_VULKAN_STEREO_MAX_TARGETS> m_encodedImports{};
|
||||
std::array<PendingTarget, kMaxImages> m_targets{};
|
||||
std::array<Import*, kMaxImages> m_encodedImports{};
|
||||
// The slots of m_targets this frame copies into, eyes first.
|
||||
std::array<uint32_t, kMaxImages> m_images{};
|
||||
uint32_t m_imageCount = 0;
|
||||
wgpu::TextureFormat m_auroraFormat = wgpu::TextureFormat::Undefined;
|
||||
AuroraVulkanStereoSubmittedCallback m_callback = nullptr;
|
||||
void* m_userdata = nullptr;
|
||||
@@ -575,7 +613,13 @@ bool aurora_vulkan_set_stereo_targets(uint64_t frameToken,
|
||||
const AuroraVulkanStereoTarget* targets,
|
||||
uint32_t targetCount) {
|
||||
using namespace aurora::vulkan_interop;
|
||||
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount);
|
||||
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, nullptr);
|
||||
}
|
||||
|
||||
bool aurora_vulkan_set_stereo_targets_with_panel(uint64_t frameToken, const AuroraVulkanStereoTarget* targets,
|
||||
uint32_t targetCount, const AuroraVulkanStereoTarget* panel) {
|
||||
using namespace aurora::vulkan_interop;
|
||||
return g_bridge && g_bridge->SetTargets(frameToken, targets, targetCount, panel);
|
||||
}
|
||||
|
||||
bool aurora_vulkan_cancel_stereo_targets(uint64_t frameToken) {
|
||||
@@ -615,6 +659,11 @@ bool aurora_vulkan_set_stereo_targets(uint64_t, const AuroraVulkanStereoTarget*,
|
||||
return false;
|
||||
}
|
||||
|
||||
bool aurora_vulkan_set_stereo_targets_with_panel(uint64_t, const AuroraVulkanStereoTarget*, uint32_t,
|
||||
const AuroraVulkanStereoTarget*) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool aurora_vulkan_cancel_stereo_targets(uint64_t) { return false; }
|
||||
|
||||
bool aurora_vulkan_disable_stereo_bridge() { return true; }
|
||||
|
||||
@@ -0,0 +1,220 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
#include <aurora/vulkan_win32_interop.h>
|
||||
#include "../internal.hpp"
|
||||
#include "../stereo.hpp"
|
||||
#include "../stereo_overlay.hpp"
|
||||
#include "gpu.hpp"
|
||||
#if defined(_WIN32) && defined(WEBGPU_DAWN) && defined(DAWN_ENABLE_BACKEND_VULKAN)
|
||||
#include <windows.h>
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <vector>
|
||||
namespace aurora::vulkan_win32 {
|
||||
namespace {
|
||||
Module Log("aurora::vulkan_win32");
|
||||
struct Api {
|
||||
AuroraDawnVulkanConfigureFn configure = nullptr;
|
||||
AuroraDawnVulkanHandlesFn handles = nullptr;
|
||||
AuroraDawnVulkanWrapFn wrap = nullptr;
|
||||
AuroraDawnVulkanReleaseFn release = nullptr;
|
||||
AuroraDawnVulkanLockFn lock = nullptr;
|
||||
AuroraDawnVulkanUnlockFn unlock = nullptr;
|
||||
AuroraDawnVulkanDrainFn drain = nullptr;
|
||||
bool Load() {
|
||||
HMODULE module = GetModuleHandleW(L"webgpu_dawn.dll");
|
||||
if (!module) return false;
|
||||
auto version = reinterpret_cast<AuroraDawnVulkanVersionFn>(GetProcAddress(module, "AuroraDawnVulkanVersion"));
|
||||
if (!version || version() != AURORA_DAWN_VULKAN_ABI) return false;
|
||||
#define LOAD(member, type, name) member = reinterpret_cast<type>(GetProcAddress(module, name)); if (!member) return false
|
||||
LOAD(configure, AuroraDawnVulkanConfigureFn, "AuroraDawnVulkanConfigure");
|
||||
LOAD(handles, AuroraDawnVulkanHandlesFn, "AuroraDawnVulkanGetHandles");
|
||||
LOAD(wrap, AuroraDawnVulkanWrapFn, "AuroraDawnVulkanWrap");
|
||||
LOAD(release, AuroraDawnVulkanReleaseFn, "AuroraDawnVulkanRelease");
|
||||
LOAD(lock, AuroraDawnVulkanLockFn, "AuroraDawnVulkanLock");
|
||||
LOAD(unlock, AuroraDawnVulkanUnlockFn, "AuroraDawnVulkanUnlock");
|
||||
LOAD(drain, AuroraDawnVulkanDrainFn, "AuroraDawnVulkanDrain");
|
||||
#undef LOAD
|
||||
return true;
|
||||
}
|
||||
} api;
|
||||
int64_t VkFormat(wgpu::TextureFormat format) {
|
||||
switch (format) {
|
||||
case wgpu::TextureFormat::RGBA8Unorm: return 37;
|
||||
case wgpu::TextureFormat::RGBA8UnormSrgb: return 43;
|
||||
case wgpu::TextureFormat::BGRA8Unorm: return 44;
|
||||
case wgpu::TextureFormat::BGRA8UnormSrgb: return 50;
|
||||
case wgpu::TextureFormat::RGBA16Float: return 97;
|
||||
default: return 0;
|
||||
}
|
||||
}
|
||||
bool CopyCompatible(int64_t a, int64_t b) {
|
||||
return a == b || ((a == 37 || a == 43) && (b == 37 || b == 43)) ||
|
||||
((a == 44 || a == 50) && (b == 44 || b == 50));
|
||||
}
|
||||
struct Import { uint64_t image; uint32_t width, height; wgpu::TextureFormat format; wgpu::Texture texture; };
|
||||
// The eyes, then the settings panel's layer image after them.
|
||||
constexpr uint32_t kMaxImages = 3;
|
||||
class Bridge {
|
||||
public:
|
||||
std::mutex mutex;
|
||||
std::vector<Import> imports;
|
||||
std::array<AuroraD3D12StereoTarget, kMaxImages> targets{};
|
||||
std::array<wgpu::Texture, kMaxImages> active{};
|
||||
uint64_t token = 0;
|
||||
uint32_t count = 0;
|
||||
// Images this frame copies: `count` eyes, plus the panel when `panel` is set.
|
||||
uint32_t images = 0;
|
||||
bool panel = false;
|
||||
bool encoded = false;
|
||||
AuroraD3D12StereoSubmittedCallback callback;
|
||||
void* userdata;
|
||||
Bridge(AuroraD3D12StereoSubmittedCallback cb, void* data) : callback(cb), userdata(data) {}
|
||||
bool Set(uint64_t next, const AuroraD3D12StereoTarget* data, uint32_t n, const AuroraD3D12StereoTarget* panelTarget) {
|
||||
if (!next || !data || !n || n > 2) return false;
|
||||
std::lock_guard guard(mutex);
|
||||
if (token) return false;
|
||||
const auto valid = [](const AuroraD3D12StereoTarget& target) {
|
||||
return target.resource && target.width && target.height;
|
||||
};
|
||||
for (uint32_t i = 0; i < n; ++i) {
|
||||
if (!valid(data[i])) return false;
|
||||
}
|
||||
if (panelTarget && !valid(*panelTarget)) return false;
|
||||
for (uint32_t i = 0; i < n; ++i) targets[i] = data[i];
|
||||
panel = panelTarget != nullptr;
|
||||
if (panel) targets[n] = *panelTarget;
|
||||
token = next; count = n; images = n + (panel ? 1u : 0u); return true;
|
||||
}
|
||||
bool Cancel(uint64_t wanted) {
|
||||
std::unique_lock guard(mutex, std::try_to_lock);
|
||||
if (!guard.owns_lock() || encoded || !token || token != wanted) return false;
|
||||
token = 0; return true;
|
||||
}
|
||||
bool Encode(wgpu::CommandEncoder& encoder, const stereo::SinkFrame& frame) {
|
||||
std::lock_guard guard(mutex);
|
||||
if (!token || encoded || frame.frameToken != token) return false;
|
||||
std::array<stereo::EyeImage, kMaxImages> sources{};
|
||||
for (uint32_t i = 0; i < count; ++i) sources[i] = frame.eyes[i];
|
||||
if (panel && !stereo_overlay::layer_source(encoder, targets[count].width, targets[count].height, sources[count]))
|
||||
return false;
|
||||
// Validate/import every target before recording any copy.
|
||||
for (uint32_t i = 0; i < images; ++i) {
|
||||
const auto& eye = sources[i];
|
||||
const auto& target = targets[i];
|
||||
if (!eye.texture || eye.size.width != target.width || eye.size.height != target.height ||
|
||||
!CopyCompatible(VkFormat(eye.format), target.dxgiFormat)) return false;
|
||||
const uint64_t image = reinterpret_cast<uintptr_t>(target.resource);
|
||||
auto it = std::find_if(imports.begin(), imports.end(), [&](const Import& entry) { return entry.image == image; });
|
||||
if (it == imports.end()) {
|
||||
wgpu::TextureDescriptor descriptor;
|
||||
descriptor.usage = wgpu::TextureUsage::CopyDst | wgpu::TextureUsage::RenderAttachment;
|
||||
descriptor.size = {target.width, target.height, 1};
|
||||
descriptor.format = eye.format;
|
||||
void* wrapped = api.wrap(webgpu::g_device.Get(), &descriptor, image);
|
||||
if (!wrapped) return false;
|
||||
imports.push_back({image, target.width, target.height, eye.format,
|
||||
wgpu::Texture::Acquire(static_cast<WGPUTexture>(wrapped))});
|
||||
it = imports.end() - 1;
|
||||
}
|
||||
if (it->width != target.width || it->height != target.height || it->format != eye.format) {
|
||||
// The runtime reuses VkImage handles across swapchain recreation, and
|
||||
// Aurora's eye format can change with the surface. Re-wrap rather than
|
||||
// rejecting every future frame for this image.
|
||||
imports.erase(it);
|
||||
--i;
|
||||
continue;
|
||||
}
|
||||
active[i] = it->texture;
|
||||
}
|
||||
for (uint32_t i = 0; i < images; ++i) {
|
||||
wgpu::TexelCopyTextureInfo source, destination;
|
||||
source.texture = *sources[i].texture;
|
||||
destination.texture = active[i];
|
||||
wgpu::Extent3D size{targets[i].width, targets[i].height, 1};
|
||||
encoder.CopyTextureToTexture(&source, &destination, &size);
|
||||
}
|
||||
encoded = true;
|
||||
return true;
|
||||
}
|
||||
void Submitted(const stereo::SinkFrame& frame) {
|
||||
std::lock_guard guard(mutex);
|
||||
if (!token || !encoded || token != frame.frameToken) return;
|
||||
std::array<void*, kMaxImages> textures{};
|
||||
for (uint32_t i = 0; i < images; ++i) textures[i] = active[i].Get();
|
||||
// Append the COLOR_ATTACHMENT_OPTIMAL release barriers to Dawn's queue,
|
||||
// flush them under its device guard, then allow the XR thread to release.
|
||||
const bool success = api.release(webgpu::g_device.Get(), textures.data(), images) != 0;
|
||||
const auto completed = token;
|
||||
token = 0; encoded = false;
|
||||
callback(completed, success, userdata);
|
||||
}
|
||||
};
|
||||
std::unique_ptr<Bridge> bridge;
|
||||
}
|
||||
}
|
||||
bool aurora_vulkan_win32_configure(const AuroraDawnVulkanHooks* hooks) {
|
||||
using namespace aurora::vulkan_win32;
|
||||
return api.Load() && api.configure(hooks);
|
||||
}
|
||||
bool aurora_vulkan_win32_get_handles(AuroraDawnVulkanHandles* handles, int64_t* format) {
|
||||
using namespace aurora;
|
||||
if (!webgpu::g_device || webgpu::g_backendType != wgpu::BackendType::Vulkan || !vulkan_win32::api.Load()) return false;
|
||||
*format = vulkan_win32::VkFormat(webgpu::g_graphicsConfig.surfaceConfiguration.format);
|
||||
return *format && vulkan_win32::api.handles(webgpu::g_device.Get(), handles);
|
||||
}
|
||||
bool aurora_vulkan_win32_enable(AuroraD3D12StereoSubmittedCallback cb, void* data) {
|
||||
using namespace aurora::vulkan_win32;
|
||||
if (bridge || !cb || !api.Load()) return false;
|
||||
bridge = std::make_unique<Bridge>(cb, data);
|
||||
aurora::stereo::set_sink(
|
||||
[](wgpu::CommandEncoder& encoder, const aurora::stereo::SinkFrame& frame, void* self) noexcept {
|
||||
return static_cast<Bridge*>(self)->Encode(encoder, frame);
|
||||
}, [](const aurora::stereo::SinkFrame& frame, void* self) noexcept { static_cast<Bridge*>(self)->Submitted(frame); }, bridge.get());
|
||||
return true;
|
||||
}
|
||||
bool aurora_vulkan_win32_set_targets(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count) {
|
||||
using namespace aurora::vulkan_win32;
|
||||
return bridge && bridge->Set(token, targets, count, nullptr);
|
||||
}
|
||||
bool aurora_vulkan_win32_set_targets_with_panel(uint64_t token, const AuroraD3D12StereoTarget* targets, uint32_t count,
|
||||
const AuroraD3D12StereoTarget* panel) {
|
||||
using namespace aurora::vulkan_win32;
|
||||
return bridge && bridge->Set(token, targets, count, panel);
|
||||
}
|
||||
bool aurora_vulkan_win32_cancel(uint64_t token) {
|
||||
using namespace aurora::vulkan_win32;
|
||||
return bridge && bridge->Cancel(token);
|
||||
}
|
||||
bool aurora_vulkan_win32_disable() {
|
||||
using namespace aurora::vulkan_win32;
|
||||
if (!bridge) return true;
|
||||
aurora::stereo::set_sink(nullptr, nullptr, nullptr);
|
||||
if (!api.drain(aurora::webgpu::g_device.Get())) { bridge.release(); return false; }
|
||||
bridge.reset();
|
||||
return true;
|
||||
}
|
||||
void* aurora_vulkan_win32_lock_queue() {
|
||||
return aurora::vulkan_win32::api.lock(aurora::webgpu::g_device.Get());
|
||||
}
|
||||
void aurora_vulkan_win32_unlock_queue(void* guard) { aurora::vulkan_win32::api.unlock(guard); }
|
||||
#else
|
||||
// C ABI stubs keep the runtime's OpenXR integration linkable on Windows GX
|
||||
// builds whose Dawn has no Vulkan backend; the backend then reports that the
|
||||
// bridge is unavailable and the game falls back to the desktop renderer.
|
||||
bool aurora_vulkan_win32_configure(const AuroraDawnVulkanHooks*) { return false; }
|
||||
bool aurora_vulkan_win32_get_handles(AuroraDawnVulkanHandles* handles, int64_t* format) {
|
||||
if (handles) *handles = {};
|
||||
if (format) *format = 0;
|
||||
return false;
|
||||
}
|
||||
bool aurora_vulkan_win32_enable(AuroraD3D12StereoSubmittedCallback, void*) { return false; }
|
||||
bool aurora_vulkan_win32_set_targets(uint64_t, const AuroraD3D12StereoTarget*, uint32_t) { return false; }
|
||||
bool aurora_vulkan_win32_set_targets_with_panel(uint64_t, const AuroraD3D12StereoTarget*, uint32_t,
|
||||
const AuroraD3D12StereoTarget*) { return false; }
|
||||
bool aurora_vulkan_win32_cancel(uint64_t) { return false; }
|
||||
bool aurora_vulkan_win32_disable() { return true; }
|
||||
void* aurora_vulkan_win32_lock_queue() { return nullptr; }
|
||||
void aurora_vulkan_win32_unlock_queue(void*) {}
|
||||
#endif
|
||||
@@ -0,0 +1,38 @@
|
||||
"""Apply the versioned Aurora native Vulkan ABI to the pinned Dawn source tree."""
|
||||
from pathlib import Path
|
||||
import shutil, sys
|
||||
root = Path(sys.argv[1]).resolve()
|
||||
here = Path(__file__).resolve().parent
|
||||
for name in ("aurora_vulkan_hooks.h", "aurora_vulkan_interop.inc"):
|
||||
shutil.copyfile(here / name, root / "src/dawn/native/vulkan" / name)
|
||||
shutil.copyfile(here.parent.parent / "include/aurora/dawn_vulkan_abi.h",
|
||||
root / "src/dawn/native/vulkan/aurora_dawn_vulkan_abi.h")
|
||||
p = root / "src/dawn/native/vulkan/VulkanBackend.cpp"
|
||||
s = p.read_text()
|
||||
if '#include "aurora_vulkan_interop.inc"' not in s:
|
||||
s += '\n#include "aurora_vulkan_interop.inc"\n'
|
||||
p.write_text(s)
|
||||
p = root / "src/dawn/common/DynamicLib.cpp"
|
||||
s = p.read_text()
|
||||
# LOAD_LIBRARY_SEARCH_DLL_LOAD_DIR requires an absolute filename. Vulkan's
|
||||
# system loader is also tried by bare name; preserve the restricted default
|
||||
# search directories for that case instead of failing with ERROR_INVALID_PARAMETER.
|
||||
old = 'LOAD_LIBRARY_SEARCH_DLL_LOAD_DIR | LOAD_LIBRARY_SEARCH_DEFAULT_DIRS;'
|
||||
new = '''((filename.size() > 2 && filename[1] == ':') || filename.starts_with("\\\\\\\\")
|
||||
? LOAD_LIBRARY_SEARCH_DLL_LOAD_DIR : 0) | LOAD_LIBRARY_SEARCH_DEFAULT_DIRS;'''
|
||||
if old in s:
|
||||
s = s.replace(old, new, 1)
|
||||
p.write_text(s)
|
||||
p = root / "src/dawn/native/vulkan/VulkanFunctions.cpp"
|
||||
s = p.read_text()
|
||||
if '#include "aurora_vulkan_hooks.h"' not in s:
|
||||
marker = 'namespace dawn::native::vulkan {'
|
||||
assert marker in s
|
||||
s = s.replace(marker, '#include "aurora_vulkan_hooks.h"\n\n' + marker, 1)
|
||||
for old, new in [
|
||||
('GET_GLOBAL_PROC(CreateInstance);', 'CreateInstance = AuroraCreateInstance(GetInstanceProcAddr);'),
|
||||
('GET_INSTANCE_PROC(CreateDevice);', 'CreateDevice = AuroraCreateDevice(GetInstanceProcAddr, instance);'),
|
||||
('GET_INSTANCE_PROC(EnumeratePhysicalDevices);', 'EnumeratePhysicalDevices = AuroraEnumeratePhysicalDevices(GetInstanceProcAddr, instance);')]:
|
||||
assert old in s, old
|
||||
s = s.replace(old, old + '\n ' + new, 1)
|
||||
p.write_text(s)
|
||||
@@ -0,0 +1,9 @@
|
||||
// Included only by the pinned Dawn source build.
|
||||
#pragma once
|
||||
#include "dawn/native/DawnNative.h"
|
||||
#include "aurora_dawn_vulkan_abi.h"
|
||||
namespace dawn::native::vulkan {
|
||||
PFN_vkCreateInstance AuroraCreateInstance(PFN_vkGetInstanceProcAddr proc);
|
||||
PFN_vkCreateDevice AuroraCreateDevice(PFN_vkGetInstanceProcAddr proc, VkInstance instance);
|
||||
PFN_vkEnumeratePhysicalDevices AuroraEnumeratePhysicalDevices(PFN_vkGetInstanceProcAddr proc, VkInstance instance);
|
||||
}
|
||||
@@ -0,0 +1,118 @@
|
||||
// Compiled inside VulkanBackend.cpp, using the pinned Dawn implementation.
|
||||
#include "aurora_vulkan_hooks.h"
|
||||
#include "src/dawn/native/ChainUtils.h"
|
||||
#include "src/dawn/native/vulkan/PhysicalDeviceVk.h"
|
||||
#include "src/dawn/native/vulkan/QueueVk.h"
|
||||
#include <mutex>
|
||||
namespace dawn::native::vulkan {
|
||||
namespace {
|
||||
AuroraDawnVulkanHooks auroraHooks{};
|
||||
std::recursive_mutex auroraHooksMutex;
|
||||
PFN_vkGetInstanceProcAddr auroraGetProc = nullptr;
|
||||
VkInstance auroraInstance = VK_NULL_HANDLE;
|
||||
::VkResult VKAPI_CALL AuroraCreateInstanceImpl(const VkInstanceCreateInfo* info,
|
||||
const VkAllocationCallbacks* allocator, VkInstance* instance) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
if (auroraHooks.createInstance) return static_cast<::VkResult>(auroraHooks.createInstance(
|
||||
auroraHooks.userdata, reinterpret_cast<void*>(auroraGetProc), info, allocator,
|
||||
reinterpret_cast<void**>(instance)));
|
||||
return reinterpret_cast<PFN_vkCreateInstance>(auroraGetProc(nullptr, "vkCreateInstance"))(info, allocator, instance);
|
||||
}
|
||||
::VkResult VKAPI_CALL AuroraCreateDeviceImpl(VkPhysicalDevice physical, const VkDeviceCreateInfo* info,
|
||||
const VkAllocationCallbacks* allocator, VkDevice* device) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
if (auroraHooks.createDevice) return static_cast<::VkResult>(auroraHooks.createDevice(
|
||||
auroraHooks.userdata, reinterpret_cast<void*>(auroraGetProc), physical, info, allocator,
|
||||
reinterpret_cast<void**>(device)));
|
||||
return reinterpret_cast<PFN_vkCreateDevice>(auroraGetProc(auroraInstance, "vkCreateDevice"))(physical, info, allocator, device);
|
||||
}
|
||||
::VkResult VKAPI_CALL AuroraEnumerateImpl(VkInstance instance, uint32_t* count, VkPhysicalDevice* devices) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
if (!auroraHooks.getPhysicalDevice) return reinterpret_cast<PFN_vkEnumeratePhysicalDevices>(
|
||||
auroraGetProc(instance, "vkEnumeratePhysicalDevices"))(instance, count, devices);
|
||||
void* physical = nullptr;
|
||||
const auto result = static_cast<::VkResult>(auroraHooks.getPhysicalDevice(auroraHooks.userdata, instance, &physical));
|
||||
if (result != VK_SUCCESS) return result;
|
||||
if (!devices) { *count = 1; return VK_SUCCESS; }
|
||||
if (*count == 0) return VK_INCOMPLETE;
|
||||
devices[0] = static_cast<VkPhysicalDevice>(physical);
|
||||
*count = 1;
|
||||
return VK_SUCCESS;
|
||||
}
|
||||
struct AuroraDeviceGuard { decltype(std::declval<Device*>()->GetGuard()) guard; explicit AuroraDeviceGuard(Device* device) : guard(device->GetGuard()) {} };
|
||||
}
|
||||
PFN_vkCreateInstance AuroraCreateInstance(PFN_vkGetInstanceProcAddr proc) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
if (!auroraHooks.createInstance) return reinterpret_cast<PFN_vkCreateInstance>(proc(nullptr, "vkCreateInstance"));
|
||||
auroraGetProc = proc;
|
||||
return AuroraCreateInstanceImpl;
|
||||
}
|
||||
PFN_vkCreateDevice AuroraCreateDevice(PFN_vkGetInstanceProcAddr proc, VkInstance instance) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
if (!auroraHooks.createDevice) return reinterpret_cast<PFN_vkCreateDevice>(proc(instance, "vkCreateDevice"));
|
||||
auroraGetProc = proc; auroraInstance = instance;
|
||||
return AuroraCreateDeviceImpl;
|
||||
}
|
||||
PFN_vkEnumeratePhysicalDevices AuroraEnumeratePhysicalDevices(PFN_vkGetInstanceProcAddr proc, VkInstance instance) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
if (!auroraHooks.getPhysicalDevice) return reinterpret_cast<PFN_vkEnumeratePhysicalDevices>(proc(instance, "vkEnumeratePhysicalDevices"));
|
||||
auroraGetProc = proc; auroraInstance = instance;
|
||||
return AuroraEnumerateImpl;
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT uint32_t AuroraDawnVulkanVersion() { return AURORA_DAWN_VULKAN_ABI; }
|
||||
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanConfigure(const AuroraDawnVulkanHooks* hooks) {
|
||||
std::lock_guard lock(auroraHooksMutex);
|
||||
// Called before adapter discovery, or cleared after Aurora shutdown. The
|
||||
// runtime and hook storage must outlive all device creation calls.
|
||||
auroraHooks = hooks ? *hooks : AuroraDawnVulkanHooks{};
|
||||
return 1;
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanGetHandles(void* handle, AuroraDawnVulkanHandles* out) {
|
||||
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
|
||||
auto guard = device->GetGuard();
|
||||
if (!device->HasFeature(Feature::ImplicitDeviceSynchronization)) return 0;
|
||||
*out = {device->GetVkInstance(), ToBackend(device->GetPhysicalDevice())->GetVkPhysicalDevice(),
|
||||
device->GetVkDevice(), device->GetGraphicsQueueFamily(), 0};
|
||||
return 1;
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT void* AuroraDawnVulkanWrap(void* handle, const void* desc, uint64_t image) {
|
||||
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
|
||||
auto guard = device->GetGuard();
|
||||
// Mirror DeviceBase::CreateTexture: fill the trivial frontend defaults
|
||||
// (dimension, mip and sample counts) and validate before the backend reads
|
||||
// them. An Undefined dimension would otherwise abort the first copy.
|
||||
TextureDescriptor raw = reinterpret_cast<const TextureDescriptor*>(desc)->WithTrivialFrontendDefaults();
|
||||
UnpackedPtr<TextureDescriptor> descriptor;
|
||||
if (device->ConsumedError(ValidateAndUnpack(&raw), &descriptor)) return nullptr;
|
||||
if (device->ConsumedError(ValidateTextureDescriptor(device, descriptor, AllowMultiPlanarTextureFormat::No))) return nullptr;
|
||||
auto texture = SwapChainTexture::Create(device, descriptor, VkImage::CreateFromHandle(reinterpret_cast<::VkImage>(image)));
|
||||
texture->SetIsSubresourceContentInitialized(true, texture->GetAllSubresources());
|
||||
texture->UpdateUsage(wgpu::TextureUsage::RenderAttachment, wgpu::ShaderStage::None, texture->GetAllSubresources());
|
||||
return ToAPI(ReturnToAPI(std::move(texture)));
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanRelease(void* handle, void* const* textures, uint32_t count) {
|
||||
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
|
||||
auto guard = device->GetGuard();
|
||||
auto* queue = ToBackend(device->GetQueue());
|
||||
auto* context = queue->GetPendingRecordingContext();
|
||||
for (uint32_t i = 0; i < count; ++i) {
|
||||
auto* texture = ToBackend(FromAPI(static_cast<WGPUTexture>(textures[i])));
|
||||
texture->TransitionUsageNow(context, wgpu::TextureUsage::RenderAttachment,
|
||||
wgpu::ShaderStage::None, texture->GetAllSubresources());
|
||||
}
|
||||
return !device->ConsumedError(queue->SubmitPendingCommands());
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT void* AuroraDawnVulkanLock(void* handle) {
|
||||
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
|
||||
return new AuroraDeviceGuard(device);
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT void AuroraDawnVulkanUnlock(void* guard) {
|
||||
delete static_cast<AuroraDeviceGuard*>(guard);
|
||||
}
|
||||
extern "C" DAWN_NATIVE_EXPORT int AuroraDawnVulkanDrain(void* handle) {
|
||||
auto* device = ToBackend(FromAPI(static_cast<WGPUDevice>(handle)));
|
||||
auto guard = device->GetGuard();
|
||||
if (device->ConsumedError(ToBackend(device->GetQueue())->SubmitPendingCommands())) return 0;
|
||||
return device->fn.QueueWaitIdle(ToBackend(device->GetQueue())->GetVkQueue()) == VK_SUCCESS;
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,18 @@ include(GoogleTest)
|
||||
|
||||
option(AURORA_GPU_SMOKE_TESTS "Build opt-in tests requiring a desktop GPU" OFF)
|
||||
if (AURORA_GPU_SMOKE_TESTS AND AURORA_ENABLE_GX AND WIN32)
|
||||
# Exercises the custom Dawn DLL's Aurora Vulkan ABI (patches/dawn). Run it with that
|
||||
# webgpu_dawn.dll beside the executable; the stock prebuilt Dawn lacks the exports and the
|
||||
# test reports that at startup. Vulkan headers come from the SDK or any tree on
|
||||
# AURORA_TEST_VULKAN_INCLUDE (the runtime's FetchContent copy works).
|
||||
find_path(AURORA_TEST_VULKAN_INCLUDE vulkan/vulkan.h HINTS "$ENV{VULKAN_SDK}/Include")
|
||||
if (AURORA_TEST_VULKAN_INCLUDE)
|
||||
add_executable(vulkan_native_bridge_smoke vulkan_native_bridge_smoke.cpp)
|
||||
target_include_directories(vulkan_native_bridge_smoke PRIVATE ../include "${AURORA_TEST_VULKAN_INCLUDE}")
|
||||
target_link_libraries(vulkan_native_bridge_smoke PRIVATE dawn::webgpu_dawn dawn::dawncpp_headers)
|
||||
else ()
|
||||
message(STATUS "vulkan_native_bridge_smoke skipped: no vulkan/vulkan.h (set VULKAN_SDK or AURORA_TEST_VULKAN_INCLUDE)")
|
||||
endif ()
|
||||
add_executable(stereo_frame_worker_smoke stereo_frame_worker_smoke.cpp)
|
||||
target_include_directories(stereo_frame_worker_smoke PRIVATE ../lib)
|
||||
target_link_libraries(stereo_frame_worker_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi
|
||||
@@ -15,6 +27,13 @@ if (AURORA_GPU_SMOKE_TESTS AND AURORA_ENABLE_GX AND WIN32)
|
||||
target_include_directories(efb_ram_lifetime_smoke PRIVATE ../lib)
|
||||
target_link_libraries(efb_ram_lifetime_smoke PRIVATE aurora::core aurora::gx aurora::main aurora::vi
|
||||
dawn::dawncpp_headers)
|
||||
# VR cockpit overlay (hands, synthetic wheel) against real scene depth. Standalone: it
|
||||
# defines the GPU globals itself and needs only the header.
|
||||
add_executable(cockpit_gpu_smoke cockpit_gpu_smoke.cpp)
|
||||
target_include_directories(cockpit_gpu_smoke PRIVATE ../include ../lib)
|
||||
target_compile_definitions(cockpit_gpu_smoke PRIVATE AURORA TARGET_PC WEBGPU_DAWN)
|
||||
target_link_libraries(cockpit_gpu_smoke PRIVATE fmt::fmt xxhash absl::flat_hash_map absl::btree
|
||||
dawn::webgpu_dawn dawn::dawncpp_headers TracyClient ${AURORA_SDL3_TARGET})
|
||||
endif ()
|
||||
|
||||
if (NOT TARGET gtest)
|
||||
@@ -36,6 +55,8 @@ if (AURORA_ENABLE_GX)
|
||||
stereo_replay_test.cpp
|
||||
stereo_interpolation_test.cpp
|
||||
stereo_mirror_test.cpp
|
||||
native_wheel_test.cpp
|
||||
cockpit_geometry_test.cpp
|
||||
texture_bind_group_cache_key_test.cpp
|
||||
../lib/gfx/efb_ram_encoder.cpp
|
||||
# GX API implementations (encoders)
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
// VR cockpit overlay geometry: what the synthetic wheel, handlebar and hands
|
||||
// build in the seated frame, without a GPU.
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <cstring>
|
||||
|
||||
#include "gfx/cockpit.hpp"
|
||||
|
||||
namespace {
|
||||
using aurora::gfx::cockpit::V;
|
||||
using aurora::gfx::cockpit::Vertex;
|
||||
|
||||
bool all_finite(const std::vector<Vertex>& vertices) {
|
||||
for (const auto& vertex : vertices) {
|
||||
for (float value : vertex.position) {
|
||||
uint32_t bits = 0;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
if ((bits & 0x7f800000u) == 0x7f800000u) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void set_identity(float (&matrix)[12], V translation) {
|
||||
const auto identity = aurora::gfx::cockpit::identity();
|
||||
std::memcpy(matrix, identity.data(), sizeof(matrix));
|
||||
matrix[3] = translation[0];
|
||||
matrix[7] = translation[1];
|
||||
matrix[11] = translation[2];
|
||||
}
|
||||
|
||||
class CockpitGeometry : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override { clear_meshes(); }
|
||||
void TearDown() override { clear_meshes(); }
|
||||
static void clear_meshes() {
|
||||
std::lock_guard lock(aurora::gfx::cockpit::meshMutex);
|
||||
aurora::gfx::cockpit::meshes = {};
|
||||
}
|
||||
};
|
||||
|
||||
TEST_F(CockpitGeometry, NativeWheelWithoutHandsDrawsNothing) {
|
||||
AuroraCockpit cockpit{};
|
||||
cockpit.nativeWheel = true;
|
||||
EXPECT_TRUE(aurora::gfx::cockpit::geometry(cockpit).empty());
|
||||
}
|
||||
|
||||
TEST_F(CockpitGeometry, SyntheticKartWheelSitsOnItsRim) {
|
||||
AuroraCockpit cockpit{};
|
||||
const auto vertices = aurora::gfx::cockpit::geometry(cockpit);
|
||||
ASSERT_FALSE(vertices.empty());
|
||||
ASSERT_TRUE(all_finite(vertices));
|
||||
// The rim, spokes and hub stay within the 0.18 m wheel plus its tube, around
|
||||
// the wheel centre the input side uses (steering_wheel.h).
|
||||
for (const auto& vertex : vertices) {
|
||||
const float x = vertex.position[0];
|
||||
const float y = vertex.position[1] + 0.30f;
|
||||
EXPECT_LE(std::hypot(x, y), 0.18f + 0.02f);
|
||||
EXPECT_NEAR(vertex.position[2], -0.42f, 0.04f);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(CockpitGeometry, SyntheticWheelTurnsWithTheAngle) {
|
||||
AuroraCockpit cockpit{};
|
||||
const auto straight = aurora::gfx::cockpit::geometry(cockpit);
|
||||
cockpit.wheelAngle = 0.5f;
|
||||
const auto turned = aurora::gfx::cockpit::geometry(cockpit);
|
||||
ASSERT_EQ(straight.size(), turned.size());
|
||||
bool moved = false;
|
||||
for (size_t i = 0; i < straight.size() && !moved; ++i) {
|
||||
moved = std::abs(straight[i].position[0] - turned[i].position[0]) > 1e-3f;
|
||||
}
|
||||
EXPECT_TRUE(moved);
|
||||
}
|
||||
|
||||
TEST_F(CockpitGeometry, SyntheticHandlebarFollowsItsFrame) {
|
||||
AuroraCockpit cockpit{};
|
||||
cockpit.bike = true;
|
||||
cockpit.handlebarRadius = 0.25f;
|
||||
// Bar axis along seat +X, centred 0.3 m down and 0.42 m ahead.
|
||||
const float pose[12]{1, 0, 0, 0, 0, 0, 1, -0.3f, 0, -1, 0, -0.42f};
|
||||
std::memcpy(cockpit.seatFromHandlebar, pose, sizeof(pose));
|
||||
const auto vertices = aurora::gfx::cockpit::geometry(cockpit);
|
||||
ASSERT_FALSE(vertices.empty());
|
||||
ASSERT_TRUE(all_finite(vertices));
|
||||
float minX = 1e9f;
|
||||
float maxX = -1e9f;
|
||||
for (const auto& vertex : vertices) {
|
||||
minX = std::min(minX, vertex.position[0]);
|
||||
maxX = std::max(maxX, vertex.position[0]);
|
||||
}
|
||||
EXPECT_NEAR(minX, -0.25f, 0.03f);
|
||||
EXPECT_NEAR(maxX, 0.25f, 0.03f);
|
||||
}
|
||||
|
||||
TEST_F(CockpitGeometry, TrackedHandDrawsAGloveAtItsGrip) {
|
||||
AuroraCockpit cockpit{};
|
||||
cockpit.nativeWheel = true;
|
||||
cockpit.hands[1].tracked = true;
|
||||
cockpit.hands[1].squeeze = 1.0f;
|
||||
set_identity(cockpit.hands[1].seatFromGrip, {0.2f, -0.3f, -0.4f});
|
||||
const auto vertices = aurora::gfx::cockpit::geometry(cockpit);
|
||||
ASSERT_FALSE(vertices.empty());
|
||||
ASSERT_TRUE(all_finite(vertices));
|
||||
for (const auto& vertex : vertices) {
|
||||
EXPECT_LT(std::abs(vertex.position[0] - 0.2f), 0.15f);
|
||||
EXPECT_LT(std::abs(vertex.position[1] + 0.3f), 0.15f);
|
||||
EXPECT_LT(std::abs(vertex.position[2] + 0.4f), 0.15f);
|
||||
}
|
||||
}
|
||||
|
||||
// The grip space OpenXR defines: -Z up the curled fingers' tube towards the
|
||||
// thumb, +X out of the palm. So the fingers run along Y (+Y on the right hand,
|
||||
// -Y on the left) and close towards +X, never out of the back of the hand.
|
||||
TEST_F(CockpitGeometry, GloveFingersRunAlongTheHandAndCloseIntoThePalm) {
|
||||
// Both grips carry the same orientation when the hands hold a wheel symmetrically, so the fingers
|
||||
// run along -Y on both and it is the palm side that mirrors: +X on the left hand, -X on the right.
|
||||
// Building the right hand's fingers on +Y instead pointed them at the player (PC, 2026-09-23).
|
||||
for (int side = 0; side < 2; ++side) {
|
||||
const float palmSide = side == 0 ? 1.0f : -1.0f;
|
||||
const auto build = [&](float squeeze) {
|
||||
AuroraCockpit cockpit{};
|
||||
cockpit.nativeWheel = true;
|
||||
cockpit.hands[side].tracked = true;
|
||||
cockpit.hands[side].squeeze = squeeze;
|
||||
set_identity(cockpit.hands[side].seatFromGrip, {0.0f, 0.0f, 0.0f});
|
||||
return aurora::gfx::cockpit::geometry(cockpit);
|
||||
};
|
||||
struct Extent {
|
||||
float reach = 0.0f; // furthest along the fingers
|
||||
float palm = 0.0f; // furthest towards the palm's normal
|
||||
float back = 0.0f; // furthest out of the back of the hand
|
||||
float across = 0.0f; // furthest across the knuckles
|
||||
};
|
||||
const auto measure = [&](const std::vector<Vertex>& vertices) {
|
||||
Extent e{};
|
||||
for (const auto& vertex : vertices) {
|
||||
e.reach = std::max(e.reach, -vertex.position[1]);
|
||||
e.palm = std::max(e.palm, vertex.position[0] * palmSide);
|
||||
e.back = std::min(e.back, vertex.position[0] * palmSide);
|
||||
e.across = std::max(e.across, std::abs(vertex.position[2]));
|
||||
}
|
||||
return e;
|
||||
};
|
||||
const auto open = measure(build(0.0f));
|
||||
const auto closed = measure(build(1.0f));
|
||||
EXPECT_GT(open.reach, 0.09f) << "open fingers reach along the hand, side " << side;
|
||||
EXPECT_LT(open.palm, 0.05f) << "an open hand is flat, side " << side;
|
||||
EXPECT_LT(closed.reach, open.reach - 0.02f) << "closing shortens the reach, side " << side;
|
||||
EXPECT_GT(closed.palm, open.palm + 0.02f) << "closing moves the fingers into the palm, side " << side;
|
||||
EXPECT_GT(closed.back, -0.03f) << "fingers never bend out of the back of the hand, side " << side;
|
||||
EXPECT_LT(closed.across, 0.07f) << "fingers stay across the knuckles, side " << side;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(CockpitGeometry, RuntimeFingersCurlTowardPalmForSqueezeAndWheelGrab) {
|
||||
using namespace aurora::gfx::cockpit;
|
||||
// OpenXR joint space: -Z runs toward the fingertip, +Y out of the back
|
||||
// of the hand, for BOTH hands. Mirror positions, not the curl direction.
|
||||
for (int side = 0; side < 2; ++side) {
|
||||
SCOPED_TRACE(side);
|
||||
HandMesh mesh;
|
||||
mesh.parents.fill(1);
|
||||
mesh.parents[1] = -1;
|
||||
const float rootPose[7]{0, 0, 0.70710678f, 0.70710678f, 0.12f, -0.08f, 0.03f};
|
||||
const M root = from_pose(rootPose);
|
||||
mesh.bind.fill(root);
|
||||
const int bases[]{2, 6, 11, 16, 21};
|
||||
for (int finger = 0; finger < 5; ++finger) {
|
||||
const int base = bases[finger];
|
||||
const int count = finger == 0 ? 4 : 5;
|
||||
for (int bone = 0; bone < count; ++bone) {
|
||||
M bind = identity();
|
||||
bind[3] = (side == 0 ? -1.0f : 1.0f) * (finger - 2) * 0.018f;
|
||||
bind[11] = -0.025f * (bone + 1);
|
||||
mesh.bind[base + bone] = compose(root, bind);
|
||||
mesh.parents[base + bone] = bone == 0 ? 1 : base + bone - 1;
|
||||
}
|
||||
}
|
||||
for (int j = 0; j < 26; ++j) mesh.inverseBind[j] = inverse(mesh.bind[j]);
|
||||
// A tiny triangle rigidly weighted to each joint, including each fingertip.
|
||||
for (int j = 0; j < 26; ++j) {
|
||||
for (V offset : {V{0, 0, 0}, V{0.001f, 0, 0}, V{0, 0, 0.001f}}) {
|
||||
AuroraVRHandVertex vertex{};
|
||||
const V p = point(mesh.bind[j].data(), offset);
|
||||
std::memcpy(vertex.position, p.data(), sizeof(vertex.position));
|
||||
vertex.joints[0] = j;
|
||||
vertex.weights[0] = 1;
|
||||
mesh.indices.push_back(static_cast<uint16_t>(mesh.vertices.size()));
|
||||
mesh.vertices.push_back(vertex);
|
||||
}
|
||||
}
|
||||
AuroraCockpitHand hand{};
|
||||
set_identity(hand.seatFromGrip, {0, 0, 0});
|
||||
const auto build = [&](float squeeze, bool held) {
|
||||
hand.squeeze = squeeze;
|
||||
hand.held = held;
|
||||
std::vector<Vertex> vertices;
|
||||
runtime_hand(vertices, hand, mesh);
|
||||
return vertices;
|
||||
};
|
||||
const auto open = build(0, false);
|
||||
for (int j = 0; j < 26; ++j) {
|
||||
const V bind = point(mesh.inverseBind[1].data(), point(mesh.bind[j].data(), {0, 0, 0}));
|
||||
for (int axis = 0; axis < 3; ++axis)
|
||||
EXPECT_NEAR(open[j * 3].position[axis], bind[axis] + (axis == 2 ? 0.04f : 0), 1e-6f);
|
||||
}
|
||||
for (const auto& closed : {build(0.5f, false), build(1, false), build(0, true)}) {
|
||||
ASSERT_TRUE(all_finite(closed));
|
||||
for (int tip : {5, 10, 15, 20, 25}) {
|
||||
EXPECT_LT(closed[tip * 3].position[1], open[tip * 3].position[1] - 0.005f)
|
||||
<< "fingertip must move toward palm (-Y), joint " << tip;
|
||||
EXPECT_GT(closed[tip * 3].position[2], open[tip * 3].position[2])
|
||||
<< "curl must shorten finger reach, joint " << tip;
|
||||
}
|
||||
for (int rigid : {0, 1, 6, 11, 16, 21})
|
||||
for (int axis = 0; axis < 3; ++axis)
|
||||
EXPECT_NEAR(closed[rigid * 3].position[axis], open[rigid * 3].position[axis], 1e-6f);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_F(CockpitGeometry, RuntimeHandMeshIsSkinnedWithoutNans) {
|
||||
using namespace aurora::gfx::cockpit;
|
||||
auto mesh = std::make_shared<HandMesh>();
|
||||
// A 26-joint chain, each joint 1 cm past its parent; one triangle on the tip.
|
||||
for (int j = 0; j < 26; ++j) {
|
||||
mesh->bind[j] = identity();
|
||||
mesh->bind[j][11] = -0.01f * float(j);
|
||||
mesh->inverseBind[j] = inverse(mesh->bind[j]);
|
||||
mesh->parents[j] = j - 1;
|
||||
}
|
||||
for (int i = 0; i < 3; ++i) {
|
||||
AuroraVRHandVertex vertex{};
|
||||
vertex.position[0] = 0.01f * float(i);
|
||||
vertex.position[2] = -0.25f;
|
||||
vertex.joints[0] = 25;
|
||||
vertex.joints[1] = vertex.joints[2] = vertex.joints[3] = -1;
|
||||
vertex.weights[0] = 1.0f;
|
||||
mesh->vertices.push_back(vertex);
|
||||
mesh->indices.push_back(uint16_t(i));
|
||||
}
|
||||
{
|
||||
std::lock_guard lock(meshMutex);
|
||||
meshes[0] = mesh;
|
||||
}
|
||||
AuroraCockpit cockpit{};
|
||||
cockpit.nativeWheel = true;
|
||||
cockpit.hands[0].tracked = true;
|
||||
cockpit.hands[0].held = true;
|
||||
set_identity(cockpit.hands[0].seatFromGrip, {-0.2f, -0.3f, -0.4f});
|
||||
const auto vertices = geometry(cockpit);
|
||||
ASSERT_EQ(vertices.size(), 3u) << "the runtime mesh replaces the glove";
|
||||
EXPECT_TRUE(all_finite(vertices));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -0,0 +1,168 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
// Ported from heurazy's mario-kart-wii-VR-port (GPL-3.0-or-later).
|
||||
// Renders the VR cockpit overlay on a real GPU against cleared, occluding and
|
||||
// partially occluding scene depth, forward and reversed, 1x and 4x MSAA.
|
||||
#include "../lib/gfx/cockpit.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <atomic>
|
||||
namespace aurora::webgpu { wgpu::Device g_device; wgpu::Queue g_queue; GraphicsConfig g_graphicsConfig{}; }
|
||||
std::atomic<int> errors=0;
|
||||
int main() {
|
||||
using namespace aurora;
|
||||
using namespace webgpu;
|
||||
wgpu::InstanceDescriptor id{};
|
||||
const wgpu::InstanceFeatureName timed=wgpu::InstanceFeatureName::TimedWaitAny;
|
||||
id.requiredFeatureCount=1;id.requiredFeatures=&timed;
|
||||
auto instance=wgpu::CreateInstance(&id);
|
||||
wgpu::Adapter adapter;
|
||||
wgpu::RequestAdapterOptions options{.backendType=wgpu::BackendType::D3D12};
|
||||
auto future=instance.RequestAdapter(&options,wgpu::CallbackMode::WaitAnyOnly,
|
||||
[&](wgpu::RequestAdapterStatus status,wgpu::Adapter a,wgpu::StringView message) {
|
||||
if(status==wgpu::RequestAdapterStatus::Success) adapter=std::move(a);
|
||||
else std::cerr<<std::string_view(message)<<'\n';
|
||||
});
|
||||
if(instance.WaitAny(future,5000000000)!=wgpu::WaitStatus::Success||!adapter) return 1;
|
||||
wgpu::DeviceDescriptor dd{};
|
||||
dd.SetUncapturedErrorCallback([](const wgpu::Device&,wgpu::ErrorType,wgpu::StringView message) {
|
||||
++errors;std::cerr<<std::string_view(message)<<'\n';
|
||||
});
|
||||
future=adapter.RequestDevice(&dd,wgpu::CallbackMode::WaitAnyOnly,
|
||||
[&](wgpu::RequestDeviceStatus status,wgpu::Device device,wgpu::StringView message) {
|
||||
if(status==wgpu::RequestDeviceStatus::Success) g_device=std::move(device);
|
||||
else std::cerr<<std::string_view(message)<<'\n';
|
||||
});
|
||||
if(instance.WaitAny(future,5000000000)!=wgpu::WaitStatus::Success||!g_device) return 1;
|
||||
g_queue=g_device.GetQueue();
|
||||
g_graphicsConfig.surfaceConfiguration.format=wgpu::TextureFormat::RGBA8Unorm;
|
||||
g_graphicsConfig.depthFormat=wgpu::TextureFormat::Depth32Float;
|
||||
AuroraCockpit native{}; native.nativeWheel=true;
|
||||
if(!gfx::cockpit::geometry(native).empty()) return 1;
|
||||
native.nativeWheel=false;
|
||||
if(gfx::cockpit::geometry(native).empty()) return 1;
|
||||
for(bool bike : {false,true}) for(bool original : {false,true}) for(uint32_t samples : {1u,4u})
|
||||
for(bool hud : {false,true}) for(int coverage : {0,1,2}) for(bool reversed : {false,true}) for(uint32_t eyeIndex : {0u,1u}) {
|
||||
const bool occluded=coverage==1;
|
||||
gfx::StereoReplayFrame frame{};
|
||||
frame.cockpit.unitsPerMeter=100;
|
||||
frame.cockpit.active=true;frame.cockpit.wheelAngle=0.35f;
|
||||
frame.cockpit.nativeWheel=original;
|
||||
frame.cockpit.bike=bike;frame.cockpit.handlebarRadius=0.25f;
|
||||
const float handlePose[12]{1,0,0,0, 0,0,1,-0.3f, 0,-1,0,-0.42f};
|
||||
std::memcpy(frame.cockpit.seatFromHandlebar,handlePose,sizeof(handlePose));
|
||||
for(int hand=0;hand<2;++hand) {
|
||||
auto& h=frame.cockpit.hands[hand];h.tracked=true;h.held=true;h.squeeze=1;
|
||||
auto pose=gfx::cockpit::identity();pose[3]=hand?0.18f:-0.18f;pose[7]=-0.30f;pose[11]=-0.42f;
|
||||
std::memcpy(h.seatFromGrip,pose.data(),sizeof(h.seatFromGrip));
|
||||
}
|
||||
wgpu::TextureDescriptor td{.usage=wgpu::TextureUsage::RenderAttachment|wgpu::TextureUsage::CopySrc,
|
||||
.size={512,512,1},.format=wgpu::TextureFormat::RGBA8Unorm,.sampleCount=1};
|
||||
auto output=g_device.CreateTexture(&td);
|
||||
td.sampleCount=samples;td.usage=wgpu::TextureUsage::RenderAttachment;
|
||||
auto color=g_device.CreateTexture(&td);
|
||||
td.format=wgpu::TextureFormat::Depth24PlusStencil8;auto depth=g_device.CreateTexture(&td);
|
||||
auto& eye=frame.eyes[eyeIndex];eye.target.colorView=samples==1?output.CreateView():color.CreateView();
|
||||
if(samples>1) eye.target.resolveView=output.CreateView();
|
||||
eye.target.depthFormat=td.format;eye.target.depthView=depth.CreateView();eye.target.size={512,512,1};eye.target.msaaSamples=samples;
|
||||
eye.projection.m0[0]=1;eye.projection.m1[1]=1;
|
||||
eye.projection.m0[2]=eyeIndex?0.06f:-0.06f;
|
||||
auto view=gfx::cockpit::identity();view[7]=0.20f;
|
||||
std::memcpy(frame.cockpit.eyeFromSeat[eyeIndex],view.data(),sizeof(frame.cockpit.eyeFromSeat[eyeIndex]));
|
||||
auto encoder=g_device.CreateCommandEncoder();
|
||||
const wgpu::RenderPassColorAttachment clear{.view=eye.target.colorView,.resolveTarget=eye.target.resolveView,
|
||||
.loadOp=wgpu::LoadOp::Clear,.storeOp=wgpu::StoreOp::Store,.clearValue={0.06,0.09,0.13,1}};
|
||||
const wgpu::RenderPassDepthStencilAttachment sceneDepth{.view=eye.target.depthView,
|
||||
.depthLoadOp=wgpu::LoadOp::Clear,.depthStoreOp=wgpu::StoreOp::Store,.depthClearValue=reversed?(occluded?0.8f:0.0f):(occluded?0.2f:1.0f),
|
||||
.stencilLoadOp=wgpu::LoadOp::Clear,.stencilStoreOp=wgpu::StoreOp::Store,.stencilClearValue=0};
|
||||
const wgpu::RenderPassDescriptor pd{.colorAttachmentCount=1,.colorAttachments=&clear,.depthStencilAttachment=&sceneDepth};
|
||||
auto pass=encoder.BeginRenderPass(&pd);
|
||||
if(coverage==2) {
|
||||
wgpu::ShaderSourceWGSL code{};
|
||||
code.code=R"(
|
||||
@vertex fn vs(@builtin(vertex_index) i:u32) -> @builtin(position) vec4f {
|
||||
let p=array<vec2f,6>(vec2f(0,-1),vec2f(1,-1),vec2f(0,1),vec2f(0,1),vec2f(1,-1),vec2f(1,1));
|
||||
return vec4f(p[i],0.5,1);
|
||||
}
|
||||
@fragment fn fs() -> @location(0) vec4f { return vec4f(0.06,0.09,0.13,1); }
|
||||
)";
|
||||
wgpu::ShaderModuleDescriptor md{};md.nextInChain=&code;
|
||||
auto shader=g_device.CreateShaderModule(&md);
|
||||
const wgpu::ColorTargetState colorState{.format=wgpu::TextureFormat::RGBA8Unorm};
|
||||
const wgpu::FragmentState fragment{.module=shader,.entryPoint="fs",.targetCount=1,.targets=&colorState};
|
||||
const wgpu::DepthStencilState ds{.format=eye.target.depthFormat,.depthWriteEnabled=true,.depthCompare=wgpu::CompareFunction::Always};
|
||||
wgpu::RenderPipelineDescriptor desc{};desc.vertex={.module=shader,.entryPoint="vs"};
|
||||
desc.fragment=&fragment;desc.depthStencil=&ds;desc.multisample.count=samples;
|
||||
auto wall=g_device.CreateRenderPipeline(&desc);pass.SetPipeline(wall);pass.Draw(6);
|
||||
}
|
||||
if(coverage==2) gfx::cockpit::render(encoder,frame,eyeIndex,reversed?gfx::cockpit::SceneDepth{0,2,true}:gfx::cockpit::SceneDepth{-1,-2,true},&pass);
|
||||
pass.End();
|
||||
if(coverage!=2) gfx::cockpit::render(encoder,frame,eyeIndex,reversed?gfx::cockpit::SceneDepth{0,2,true}:gfx::cockpit::SceneDepth{-1,-2,true});
|
||||
if(hud) {
|
||||
// An opaque black screen and coloured HUD with depth testing disabled used
|
||||
// to overwrite the hands. Exercise a later pass too: the mask must survive.
|
||||
const wgpu::RenderPassColorAttachment load{.view=eye.target.colorView,.resolveTarget=eye.target.resolveView,
|
||||
.loadOp=wgpu::LoadOp::Load,.storeOp=wgpu::StoreOp::Store};
|
||||
const wgpu::RenderPassDepthStencilAttachment loadDepth{.view=eye.target.depthView,
|
||||
.depthLoadOp=wgpu::LoadOp::Load,.depthStoreOp=wgpu::StoreOp::Store,
|
||||
.stencilLoadOp=wgpu::LoadOp::Load,.stencilStoreOp=wgpu::StoreOp::Store};
|
||||
const wgpu::RenderPassDescriptor hudPass{.colorAttachmentCount=1,.colorAttachments=&load,.depthStencilAttachment=&loadDepth};
|
||||
auto overlay=encoder.BeginRenderPass(&hudPass);
|
||||
wgpu::ShaderSourceWGSL code{};
|
||||
code.code=R"(
|
||||
@vertex fn vs(@builtin(vertex_index) i:u32) -> @builtin(position) vec4f {
|
||||
let p=array<vec2f,3>(vec2f(-1,-1),vec2f(3,-1),vec2f(-1,3));
|
||||
return vec4f(p[i],0.5,1);
|
||||
}
|
||||
@fragment fn fs(@builtin(position) p:vec4f) -> @location(0) vec4f {
|
||||
return select(vec4f(0,0,0,1),vec4f(1,0,0,1),p.x<256);
|
||||
}
|
||||
)";
|
||||
wgpu::ShaderModuleDescriptor md{};md.nextInChain=&code;auto shader=g_device.CreateShaderModule(&md);
|
||||
const wgpu::ColorTargetState colorState{.format=wgpu::TextureFormat::RGBA8Unorm};
|
||||
const wgpu::FragmentState fragment{.module=shader,.entryPoint="fs",.targetCount=1,.targets=&colorState};
|
||||
const wgpu::StencilFaceState mask{.compare=wgpu::CompareFunction::Equal};
|
||||
const wgpu::DepthStencilState ds{.format=eye.target.depthFormat,.depthWriteEnabled=true,
|
||||
.depthCompare=wgpu::CompareFunction::Always,.stencilFront=mask,.stencilBack=mask,.stencilReadMask=1,.stencilWriteMask=0};
|
||||
wgpu::RenderPipelineDescriptor desc{};desc.vertex={.module=shader,.entryPoint="vs"};
|
||||
desc.fragment=&fragment;desc.depthStencil=&ds;desc.multisample.count=samples;
|
||||
auto screen=g_device.CreateRenderPipeline(&desc);overlay.SetPipeline(screen);overlay.Draw(3);overlay.End();
|
||||
}
|
||||
const wgpu::BufferDescriptor bd{.usage=wgpu::BufferUsage::CopyDst|wgpu::BufferUsage::MapRead,.size=512*512*4};
|
||||
auto readback=g_device.CreateBuffer(&bd);
|
||||
const wgpu::TexelCopyTextureInfo src{.texture=output};
|
||||
const wgpu::TexelCopyBufferInfo dst{.layout={.bytesPerRow=2048,.rowsPerImage=512},.buffer=readback};
|
||||
const wgpu::Extent3D extent{512,512,1};encoder.CopyTextureToBuffer(&src,&dst,&extent);
|
||||
auto commands=encoder.Finish();g_device.GetQueue().Submit(1,&commands);
|
||||
bool mapped=false;
|
||||
future=readback.MapAsync(wgpu::MapMode::Read,0,512*512*4,wgpu::CallbackMode::WaitAnyOnly,
|
||||
[&](wgpu::MapAsyncStatus status,wgpu::StringView) { mapped=status==wgpu::MapAsyncStatus::Success; });
|
||||
if(instance.WaitAny(future,5000000000)!=wgpu::WaitStatus::Success||!mapped) return 1;
|
||||
const auto* bytes=static_cast<const unsigned char*>(readback.GetConstMappedRange());
|
||||
size_t bright=0;
|
||||
for(size_t i=0;i<512*512;++i) if(bytes[4*i]>90&&bytes[4*i+1]>90&&bytes[4*i+2]>90) ++bright;
|
||||
if(hud && !(bytes[0]>250 && bytes[1]==0 && bytes[2]==0)) {
|
||||
std::cerr<<"HUD missing outside cockpit mask\n";++errors;
|
||||
}
|
||||
static size_t expectedBright[3][2][2]{};
|
||||
auto& expected=expectedBright[coverage][reversed][eyeIndex];
|
||||
if(!hud) expected=bright;
|
||||
else if(bright+64<expected || bright>expected+64) { std::cerr<<"HUD changed visible cockpit pixels\n";++errors; }
|
||||
if(occluded ? bright!=0 : bright<1000) { std::cerr<<"Incorrect hands/wheel occlusion\n";++errors; }
|
||||
if(coverage==2) {
|
||||
size_t left=0,right=0;
|
||||
for(size_t y=0;y<512;++y) for(size_t x=0;x<512;++x) {
|
||||
const auto i=y*512+x;
|
||||
if(bytes[4*i]>90&&bytes[4*i+1]>90&&bytes[4*i+2]>90) (x<256?left:right)++;
|
||||
}
|
||||
if(left<500||right>8) { std::cerr<<"Partial wall occlusion failed for eye "<<eyeIndex<<'\n';++errors; }
|
||||
}
|
||||
if(samples==4 && !original && !occluded) {
|
||||
std::ofstream image("cockpit-preview.ppm",std::ios::binary);image<<"P6\n512 512\n255\n";
|
||||
for(size_t i=0;i<512*512;++i) image.write(reinterpret_cast<const char*>(bytes+i*4),3);
|
||||
}
|
||||
readback.Unmap();
|
||||
std::cout<<(bike?"Bike ":"Kart ")<<(original?"native hands: ":"VR controls: ")<<samples<<"x MSAA: "<<bright<<" visible geometry pixels\n";
|
||||
}
|
||||
gfx::cockpit::shutdown();g_queue=nullptr;g_device.Destroy();g_device=nullptr;
|
||||
return errors?1:0;
|
||||
}
|
||||
@@ -0,0 +1,207 @@
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
// Native steering wheel: indexed-matrix ownership and array matching. The
|
||||
// ownership cases come from heurazy's mario-kart-wii-VR-port test set.
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include <array>
|
||||
#include <vector>
|
||||
|
||||
#include "gx_test_common.hpp"
|
||||
#include "gx/native_wheel.hpp"
|
||||
|
||||
namespace {
|
||||
|
||||
// Three positions, 12 bytes each. Position 1 is on the wheel; 0 and 2 belong to
|
||||
// the body/wing.
|
||||
struct OwnershipFixture {
|
||||
std::array<uint8_t, 36> original{};
|
||||
std::array<uint8_t, 36> replacement{};
|
||||
// PN matrix byte, texture matrix byte, big-endian position index.
|
||||
std::array<uint8_t, 12> vertices{0, 30, 0, 0, 0, 30, 0, 1, 3, 30, 0, 2};
|
||||
OwnershipFixture() { replacement[12] = 7; }
|
||||
bool matches(uint16_t mask) const {
|
||||
return aurora::NativeWheelDrawMatches(original, replacement, 12, vertices, 4, 2, 2, mask);
|
||||
}
|
||||
};
|
||||
|
||||
TEST(NativeWheelMatch, MixedBodyAndWingDrawAcceptsWheelPositionsOnLocalBody) {
|
||||
OwnershipFixture f;
|
||||
EXPECT_TRUE(f.matches(1));
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, UnrelatedWingJointCannotAnimateWheel) {
|
||||
OwnershipFixture f;
|
||||
EXPECT_FALSE(f.matches(2));
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, OpponentWithoutLocalMatrixIsRejected) {
|
||||
OwnershipFixture f;
|
||||
EXPECT_FALSE(f.matches(0));
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, WheelPositionUnderAnotherMatrixIsRejected) {
|
||||
OwnershipFixture f;
|
||||
f.vertices[4] = 6;
|
||||
EXPECT_FALSE(f.matches(1));
|
||||
EXPECT_TRUE(f.matches(4)) << "local body can occupy a different palette slot";
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, MalformedMatrixSelectorRejected) {
|
||||
OwnershipFixture f;
|
||||
f.vertices[4] = 1;
|
||||
EXPECT_FALSE(f.matches(1));
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, OutOfRangePositionIndexRejected) {
|
||||
OwnershipFixture f;
|
||||
f.vertices[11] = 3;
|
||||
EXPECT_FALSE(f.matches(1));
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, SelectionTouchingAnotherJointCannotDeformIt) {
|
||||
OwnershipFixture f;
|
||||
f.replacement[24] = 1;
|
||||
EXPECT_FALSE(f.matches(1));
|
||||
}
|
||||
|
||||
TEST(NativeWheelMatch, EightBitIndicesAndLayoutValidation) {
|
||||
OwnershipFixture f;
|
||||
std::array<uint8_t, 4> indexed8{0, 1, 3, 2};
|
||||
EXPECT_TRUE(aurora::NativeWheelDrawMatches(f.original, f.replacement, 12, indexed8, 2, 1, 1, 1));
|
||||
EXPECT_FALSE(aurora::NativeWheelDrawMatches(f.original, f.replacement, 12, indexed8, 2, 2, 1, 1))
|
||||
<< "invalid vertex layout rejected";
|
||||
EXPECT_FALSE(aurora::NativeWheelDrawMatches(f.original, f.original, 12, indexed8, 2, 1, 1, 1))
|
||||
<< "unchanged positions are not counted as animation";
|
||||
}
|
||||
|
||||
// Draw-time matching against the GX position-matrix palette.
|
||||
class NativeWheelArrayTest : public ::testing::Test {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
aurora::gx::g_gxState = {};
|
||||
aurora::gx::nativeWheelArrays.clear();
|
||||
aurora::gx::nativeWheelMatches = 0;
|
||||
aurora::gx::NativeWheelArray replacement;
|
||||
replacement.source = source.data();
|
||||
replacement.bytes.assign(36, 0);
|
||||
replacement.bytes[12] = 1;
|
||||
aurora::gx::nativeWheelArrays.push_back(replacement);
|
||||
array.data = source.data();
|
||||
array.size = 36;
|
||||
array.stride = 12;
|
||||
}
|
||||
void TearDown() override {
|
||||
aurora::gx::g_gxState = {};
|
||||
aurora::gx::nativeWheelArrays.clear();
|
||||
}
|
||||
static float* pos(uint32_t slot) { return reinterpret_cast<float*>(&aurora::gx::g_gxState.pnMtx[slot].pos); }
|
||||
std::array<uint8_t, 36> source{};
|
||||
aurora::gx::AttrArray array{};
|
||||
};
|
||||
|
||||
TEST_F(NativeWheelArrayTest, MatchesLocalVehicleMatrixOnly) {
|
||||
EXPECT_NE(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr);
|
||||
EXPECT_EQ(aurora::gx::nativeWheelMatches, 1u);
|
||||
// Same asset drawn by an opponent: a translated matrix.
|
||||
pos(0)[3] = 100;
|
||||
EXPECT_EQ(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr);
|
||||
EXPECT_EQ(aurora::gx::nativeWheelMatches, 1u);
|
||||
}
|
||||
|
||||
TEST_F(NativeWheelArrayTest, UnboundSourceIsNotAWheelArray) {
|
||||
std::array<uint8_t, 36> other{};
|
||||
aurora::gx::AttrArray unrelated = array;
|
||||
unrelated.data = other.data();
|
||||
EXPECT_EQ(aurora::gx::native_wheel_array(unrelated, nullptr, 0, 0, 0), nullptr);
|
||||
EXPECT_TRUE(aurora::gx::native_wheel_source(source.data()));
|
||||
EXPECT_FALSE(aurora::gx::native_wheel_source(other.data()));
|
||||
}
|
||||
|
||||
TEST_F(NativeWheelArrayTest, MultiJointBodyChecksPerPositionOwnership) {
|
||||
// Multi-joint body with an independently animated wing, as in Mario's
|
||||
// mb/mc kart models. Only the wheel position uses the matching body slot.
|
||||
aurora::gx::g_gxState.vtxDesc[GX_VA_PNMTXIDX] = GX_DIRECT;
|
||||
aurora::gx::g_gxState.vtxDesc[GX_VA_POS] = GX_INDEX16;
|
||||
for (uint32_t slot = 0; slot < aurora::gx::MaxPnMtx; ++slot) pos(slot)[3] = 100;
|
||||
pos(2)[3] = 0;
|
||||
uint8_t vertices[]{6, 0, 1, 3, 0, 2};
|
||||
EXPECT_NE(aurora::gx::native_wheel_array(array, vertices, sizeof(vertices), 3, 1), nullptr);
|
||||
vertices[0] = 0; // opponent wheel while an unrelated palette slot still matches
|
||||
EXPECT_EQ(aurora::gx::native_wheel_array(array, vertices, sizeof(vertices), 3, 1), nullptr);
|
||||
}
|
||||
|
||||
TEST_F(NativeWheelArrayTest, NonFiniteMatrixNeverMatches) {
|
||||
pos(0)[0] = __builtin_nanf("");
|
||||
EXPECT_EQ(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr);
|
||||
}
|
||||
|
||||
// Draw merging around an animated array. A merged draw appends its vertices to the previous draw's range and renders
|
||||
// through that draw's array binding, so two draws may merge only when they resolved the same replacement. The kart's
|
||||
// display list is hundreds of same-state primitives, and merging them is worth several ms an eye on a tiler.
|
||||
class NativeWheelMergeTest : public GXFifoTest {
|
||||
protected:
|
||||
void SetUp() override {
|
||||
GXFifoTest::SetUp();
|
||||
aurora::gfx::testing::use_draw_command_tracking(true);
|
||||
aurora::gx::nativeWheelArrays.clear();
|
||||
aurora::gx::nativeWheelLastDecision = nullptr;
|
||||
aurora::gx::nativeWheelLastDrawCommand = nullptr;
|
||||
aurora::gx::NativeWheelArray replacement;
|
||||
replacement.source = source.data();
|
||||
replacement.bytes.assign(source.size(), 0);
|
||||
replacement.bytes[12] = 1; // one animated position
|
||||
aurora::gx::nativeWheelArrays.push_back(replacement);
|
||||
auto& state = aurora::gx::g_gxState;
|
||||
state.lastVtxFmt = GX_VTXFMT0;
|
||||
state.lastVtxSize = 1;
|
||||
state.vtxDesc[GX_VA_POS] = GX_INDEX8;
|
||||
state.arrays[GX_VA_POS].data = source.data();
|
||||
state.arrays[GX_VA_POS].size = static_cast<u32>(source.size());
|
||||
state.arrays[GX_VA_POS].stride = 12;
|
||||
state.stateDirty = true;
|
||||
}
|
||||
void TearDown() override {
|
||||
aurora::gx::nativeWheelArrays.clear();
|
||||
aurora::gx::nativeWheelLastDecision = nullptr;
|
||||
aurora::gx::nativeWheelLastDrawCommand = nullptr;
|
||||
}
|
||||
void draw() {
|
||||
std::vector<u8> fifo{static_cast<u8>(GX_TRIANGLES) | static_cast<u8>(GX_VTXFMT0), 0, 3, 0, 1, 2};
|
||||
decode_fifo(fifo);
|
||||
}
|
||||
// The local vehicle's matrix: the replacement's model-view is all zeroes, and so is a default palette slot.
|
||||
void makeOpponent() { reinterpret_cast<float*>(&aurora::gx::g_gxState.pnMtx[0].pos)[3] = 100.f; }
|
||||
std::array<uint8_t, 36> source{};
|
||||
};
|
||||
|
||||
TEST_F(NativeWheelMergeTest, PrimitivesSharingTheAnimatedArrayStillMerge) {
|
||||
draw();
|
||||
ASSERT_EQ(aurora::gx::nativeWheelLastDecision, &aurora::gx::nativeWheelArrays.front());
|
||||
draw();
|
||||
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u);
|
||||
}
|
||||
|
||||
TEST_F(NativeWheelMergeTest, PrimitivesThatTakeTheOriginalArrayAlsoStillMerge) {
|
||||
makeOpponent();
|
||||
draw();
|
||||
ASSERT_EQ(aurora::gx::nativeWheelLastDecision, nullptr);
|
||||
draw();
|
||||
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u);
|
||||
}
|
||||
|
||||
TEST_F(NativeWheelMergeTest, OpponentPrimitiveNeverFoldsIntoAnAnimatedDraw) {
|
||||
draw();
|
||||
makeOpponent();
|
||||
draw();
|
||||
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u) << "the merged whole would render the opponent animated";
|
||||
}
|
||||
|
||||
TEST_F(NativeWheelMergeTest, AnimatedPrimitiveNeverFoldsIntoAnOpponentDraw) {
|
||||
makeOpponent();
|
||||
draw();
|
||||
reinterpret_cast<float*>(&aurora::gx::g_gxState.pnMtx[0].pos)[3] = 0.f;
|
||||
draw();
|
||||
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u) << "the wheel would render on the original vertices";
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -0,0 +1,162 @@
|
||||
// Opt-in real GPU test for the custom Dawn DLL. No headset required.
|
||||
// Exercises the MinGW/MSVC C ABI, borrowed-image lifetime and repeated layout
|
||||
// transitions. Run with VK_INSTANCE_LAYERS=VK_LAYER_KHRONOS_validation as well.
|
||||
#define NOMINMAX
|
||||
#include <windows.h>
|
||||
#include <vulkan/vulkan.h>
|
||||
#include <dawn/webgpu_cpp.h>
|
||||
#include <aurora/dawn_vulkan_abi.h>
|
||||
#include <array>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
|
||||
static void Check(bool ok) { if (!ok) { std::fputs("Native Vulkan bridge smoke failed\n", stderr); std::abort(); } }
|
||||
static PFN_vkGetInstanceProcAddr hookProc;
|
||||
static VkInstance hookInstance;
|
||||
static int instances = 0, devices = 0, selections = 0;
|
||||
static int32_t CreateInstance(void*, void* proc, const void* info, const void* allocator, void** out) {
|
||||
++instances; hookProc = reinterpret_cast<PFN_vkGetInstanceProcAddr>(proc);
|
||||
auto create = reinterpret_cast<PFN_vkCreateInstance>(hookProc(nullptr, "vkCreateInstance"));
|
||||
auto result = create(static_cast<const VkInstanceCreateInfo*>(info), static_cast<const VkAllocationCallbacks*>(allocator), &hookInstance);
|
||||
*out = hookInstance; return result;
|
||||
}
|
||||
static int32_t CreateDevice(void*, void*, void* physical, const void* info, const void* allocator, void** out) {
|
||||
++devices;
|
||||
auto create = reinterpret_cast<PFN_vkCreateDevice>(hookProc(hookInstance, "vkCreateDevice"));
|
||||
return create(static_cast<VkPhysicalDevice>(physical), static_cast<const VkDeviceCreateInfo*>(info),
|
||||
static_cast<const VkAllocationCallbacks*>(allocator), reinterpret_cast<VkDevice*>(out));
|
||||
}
|
||||
static int32_t SelectPhysical(void*, void* instance, void** out) {
|
||||
++selections;
|
||||
auto enumerate = reinterpret_cast<PFN_vkEnumeratePhysicalDevices>(hookProc(static_cast<VkInstance>(instance), "vkEnumeratePhysicalDevices"));
|
||||
uint32_t count = 1;
|
||||
auto result = enumerate(static_cast<VkInstance>(instance), &count, reinterpret_cast<VkPhysicalDevice*>(out));
|
||||
return result == VK_INCOMPLETE ? VK_SUCCESS : result;
|
||||
}
|
||||
int main() {
|
||||
auto dll = LoadLibraryW(L"webgpu_dawn.dll"); Check(dll != nullptr);
|
||||
#define API(name, type) auto name = reinterpret_cast<type>(GetProcAddress(dll, "AuroraDawnVulkan" #name)); Check(name != nullptr)
|
||||
API(Version, AuroraDawnVulkanVersionFn);
|
||||
API(Configure, AuroraDawnVulkanConfigureFn);
|
||||
API(GetHandles, AuroraDawnVulkanHandlesFn);
|
||||
API(Wrap, AuroraDawnVulkanWrapFn);
|
||||
API(Release, AuroraDawnVulkanReleaseFn);
|
||||
API(Lock, AuroraDawnVulkanLockFn);
|
||||
API(Unlock, AuroraDawnVulkanUnlockFn);
|
||||
API(Drain, AuroraDawnVulkanDrainFn);
|
||||
Check(Version() == AURORA_DAWN_VULKAN_ABI);
|
||||
AuroraDawnVulkanHooks hooks{nullptr, CreateInstance, CreateDevice, SelectPhysical};
|
||||
Check(Configure(&hooks));
|
||||
wgpu::InstanceFeatureName feature = wgpu::InstanceFeatureName::TimedWaitAny;
|
||||
wgpu::InstanceDescriptor instanceDesc;
|
||||
instanceDesc.requiredFeatureCount = 1; instanceDesc.requiredFeatures = &feature;
|
||||
std::fputs("Creating WebGPU instance\n", stderr);
|
||||
auto instance = wgpu::CreateInstance(&instanceDesc);
|
||||
wgpu::RequestAdapterOptions options; options.backendType = wgpu::BackendType::Vulkan;
|
||||
wgpu::Adapter adapter;
|
||||
std::fputs("Requesting Vulkan adapter\n", stderr);
|
||||
auto future = instance.RequestAdapter(&options, wgpu::CallbackMode::WaitAnyOnly,
|
||||
[&](wgpu::RequestAdapterStatus status, wgpu::Adapter value, wgpu::StringView) {
|
||||
Check(status == wgpu::RequestAdapterStatus::Success); adapter = std::move(value);
|
||||
});
|
||||
Check(instance.WaitAny(future, 10'000'000'000) == wgpu::WaitStatus::Success);
|
||||
wgpu::FeatureName sync = wgpu::FeatureName::ImplicitDeviceSynchronization;
|
||||
wgpu::DeviceDescriptor deviceDesc;
|
||||
deviceDesc.requiredFeatureCount = 1; deviceDesc.requiredFeatures = &sync;
|
||||
deviceDesc.SetUncapturedErrorCallback([](const wgpu::Device&, wgpu::ErrorType, wgpu::StringView message) {
|
||||
std::fprintf(stderr, "Dawn: %.*s\n", static_cast<int>(message.length), message.data); Check(false);
|
||||
});
|
||||
std::fputs("Creating Vulkan device\n", stderr);
|
||||
auto device = adapter.CreateDevice(&deviceDesc); Check(!!device);
|
||||
Check(instances > 0 && devices > 0 && selections > 0);
|
||||
std::fputs("Getting native device\n", stderr);
|
||||
AuroraDawnVulkanHandles handles{}; Check(GetHandles(device.Get(), &handles));
|
||||
VkDevice vkDevice = static_cast<VkDevice>(handles.device);
|
||||
VkPhysicalDevice physical = static_cast<VkPhysicalDevice>(handles.physicalDevice);
|
||||
auto loader = LoadLibraryW(L"vulkan-1.dll"); Check(loader != nullptr);
|
||||
auto getProc = reinterpret_cast<PFN_vkGetInstanceProcAddr>(GetProcAddress(loader, "vkGetInstanceProcAddr"));
|
||||
#define VK(name) auto name = reinterpret_cast<PFN_##name>(getProc(static_cast<VkInstance>(handles.instance), #name)); Check(name != nullptr)
|
||||
VK(vkCreateImage); VK(vkGetImageMemoryRequirements); VK(vkGetPhysicalDeviceMemoryProperties);
|
||||
VK(vkAllocateMemory); VK(vkBindImageMemory); VK(vkCreateCommandPool); VK(vkAllocateCommandBuffers);
|
||||
VK(vkBeginCommandBuffer); VK(vkCmdPipelineBarrier); VK(vkEndCommandBuffer); VK(vkGetDeviceQueue);
|
||||
VK(vkQueueSubmit); VK(vkQueueWaitIdle); VK(vkDestroyCommandPool); VK(vkDestroyImage); VK(vkFreeMemory);
|
||||
VkImageCreateInfo imageInfo{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
|
||||
imageInfo.imageType = VK_IMAGE_TYPE_2D; imageInfo.format = VK_FORMAT_R8G8B8A8_UNORM;
|
||||
imageInfo.extent = {64, 16, 1}; imageInfo.mipLevels = 1; imageInfo.arrayLayers = 1;
|
||||
imageInfo.samples = VK_SAMPLE_COUNT_1_BIT; imageInfo.tiling = VK_IMAGE_TILING_OPTIMAL;
|
||||
imageInfo.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
|
||||
std::fputs("Creating borrowed Vulkan image\n", stderr);
|
||||
VkImage image; Check(vkCreateImage(vkDevice, &imageInfo, nullptr, &image) == VK_SUCCESS);
|
||||
VkMemoryRequirements requirements; vkGetImageMemoryRequirements(vkDevice, image, &requirements);
|
||||
VkPhysicalDeviceMemoryProperties properties; vkGetPhysicalDeviceMemoryProperties(physical, &properties);
|
||||
uint32_t memoryType = 0;
|
||||
while (!(requirements.memoryTypeBits & (1u << memoryType))) ++memoryType;
|
||||
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
||||
allocation.allocationSize = requirements.size; allocation.memoryTypeIndex = memoryType;
|
||||
VkDeviceMemory memory; Check(vkAllocateMemory(vkDevice, &allocation, nullptr, &memory) == VK_SUCCESS);
|
||||
Check(vkBindImageMemory(vkDevice, image, memory, 0) == VK_SUCCESS);
|
||||
VkCommandPoolCreateInfo poolInfo{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; poolInfo.queueFamilyIndex = handles.queueFamily;
|
||||
VkCommandPool pool; Check(vkCreateCommandPool(vkDevice, &poolInfo, nullptr, &pool) == VK_SUCCESS);
|
||||
VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
|
||||
alloc.commandPool = pool; alloc.commandBufferCount = 1; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY;
|
||||
VkCommandBuffer commands; Check(vkAllocateCommandBuffers(vkDevice, &alloc, &commands) == VK_SUCCESS);
|
||||
VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
|
||||
Check(vkBeginCommandBuffer(commands, &begin) == VK_SUCCESS);
|
||||
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
|
||||
barrier.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; barrier.newLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL;
|
||||
barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
||||
barrier.image = image; barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
|
||||
barrier.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
|
||||
vkCmdPipelineBarrier(commands, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT,
|
||||
0, 0, nullptr, 0, nullptr, 1, &barrier);
|
||||
Check(vkEndCommandBuffer(commands) == VK_SUCCESS);
|
||||
VkQueue queue; vkGetDeviceQueue(vkDevice, handles.queueFamily, handles.queueIndex, &queue);
|
||||
VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; submit.commandBufferCount = 1; submit.pCommandBuffers = &commands;
|
||||
std::fputs("Locking native queue\n", stderr);
|
||||
auto guard = Lock(device.Get());
|
||||
Check(vkQueueSubmit(queue, 1, &submit, VK_NULL_HANDLE) == VK_SUCCESS);
|
||||
Check(vkQueueWaitIdle(queue) == VK_SUCCESS); Unlock(guard);
|
||||
wgpu::TextureDescriptor textureDesc;
|
||||
textureDesc.size = {64, 16, 1}; textureDesc.format = wgpu::TextureFormat::RGBA8Unorm;
|
||||
textureDesc.usage = wgpu::TextureUsage::CopyDst | wgpu::TextureUsage::CopySrc | wgpu::TextureUsage::RenderAttachment;
|
||||
std::fputs("Wrapping borrowed image\n", stderr);
|
||||
auto borrowed = wgpu::Texture::Acquire(static_cast<WGPUTexture>(Wrap(device.Get(), &textureDesc, reinterpret_cast<uint64_t>(image))));
|
||||
Check(!!borrowed);
|
||||
std::fputs("Creating copy resources\n", stderr);
|
||||
auto source = device.CreateTexture(&textureDesc);
|
||||
wgpu::BufferDescriptor bufferDesc; bufferDesc.size = 4096;
|
||||
bufferDesc.usage = wgpu::BufferUsage::CopyDst | wgpu::BufferUsage::MapRead;
|
||||
auto readback = device.CreateBuffer(&bufferDesc);
|
||||
std::fputs("Running GPU copy\n", stderr);
|
||||
for (uint8_t value : {17, 99, 201}) {
|
||||
std::array<uint8_t, 4096> pixels; pixels.fill(value);
|
||||
wgpu::TexelCopyTextureInfo sourceInfo; sourceInfo.texture = source;
|
||||
wgpu::TexelCopyTextureInfo targetInfo; targetInfo.texture = borrowed;
|
||||
wgpu::TexelCopyBufferLayout layout; layout.bytesPerRow = 256; layout.rowsPerImage = 16;
|
||||
wgpu::Extent3D extent{64, 16, 1};
|
||||
device.GetQueue().WriteTexture(&sourceInfo, pixels.data(), pixels.size(), &layout, &extent);
|
||||
std::fputs("WriteTexture done\n", stderr);
|
||||
auto encoder = device.CreateCommandEncoder(); encoder.CopyTextureToTexture(&sourceInfo, &targetInfo, &extent);
|
||||
auto copy = encoder.Finish(); device.GetQueue().Submit(1, ©);
|
||||
std::fputs("Copy submitted\n", stderr);
|
||||
void* textures[] = {borrowed.Get()}; Check(Release(device.Get(), textures, 1));
|
||||
std::fputs("Release transition done\n", stderr);
|
||||
encoder = device.CreateCommandEncoder();
|
||||
wgpu::TexelCopyBufferInfo destination; destination.buffer = readback; destination.layout = layout;
|
||||
encoder.CopyTextureToBuffer(&targetInfo, &destination, &extent);
|
||||
copy = encoder.Finish(); device.GetQueue().Submit(1, ©);
|
||||
Check(Release(device.Get(), textures, 1));
|
||||
auto mapped = readback.MapAsync(wgpu::MapMode::Read, 0, 4096, wgpu::CallbackMode::WaitAnyOnly,
|
||||
[](wgpu::MapAsyncStatus status, wgpu::StringView) { Check(status == wgpu::MapAsyncStatus::Success); });
|
||||
Check(instance.WaitAny(mapped, 10'000'000'000) == wgpu::WaitStatus::Success);
|
||||
const auto* bytes = static_cast<const uint8_t*>(readback.GetConstMappedRange());
|
||||
for (size_t i = 0; i < pixels.size(); ++i) Check(bytes[i] == value);
|
||||
readback.Unmap();
|
||||
}
|
||||
Check(Drain(device.Get())); borrowed = nullptr;
|
||||
// The wrapper must not free the runtime-owned image or its memory.
|
||||
vkDestroyImage(vkDevice, image, nullptr); vkFreeMemory(vkDevice, memory, nullptr);
|
||||
vkDestroyCommandPool(vkDevice, pool, nullptr);
|
||||
Check(Configure(nullptr));
|
||||
std::puts("Native Vulkan bridge: three GPU copy/readback cycles passed");
|
||||
}
|
||||
Reference in new issue
Block a user