#pragma once #include "../internal.hpp" #include "../webgpu/gpu.hpp" #include #include #include #include #include #include #include #include #include #include #include #include #define XXH_STATIC_LINKING_ONLY #include namespace aurora { #if INTPTR_MAX == INT32_MAX using HashType = XXH32_hash_t; #else using HashType = XXH64_hash_t; #endif static inline HashType xxh3_hash_s(const void* input, size_t len, HashType seed = 0) { return static_cast(XXH3_64bits_withSeed(input, len, seed)); } template static inline HashType xxh3_hash(const T& input, HashType seed = 0) { // Validate that the type has no padding bytes, which can easily cause // hash mismatches. This also disallows floats, but that's okay for us. static_assert(std::has_unique_object_representations_v); return xxh3_hash_s(&input, sizeof(T), seed); } // Folds two 64-bit hashes. Chaining them instead drops XXH3 onto its seeded long path past 240 // bytes, and that path regenerates a 192-byte secret on every call. static inline HashType hash_combine(HashType lhs, HashType rhs) { uint64_t mixed = static_cast(lhs) ^ (static_cast(rhs) + 0x9E3779B97F4A7C15ull + (static_cast(lhs) << 6) + (static_cast(lhs) >> 2)); mixed ^= mixed >> 33; mixed *= 0xFF51AFD7ED558CCDull; mixed ^= mixed >> 29; return static_cast(mixed); } // Guest-RAM write tracking hooks (see aurora_set_guest_write_hooks). A digest samples the // generation before reading the bytes, so a racing write costs an extra digest, never a skipped one. inline constexpr uint64_t kGuestWriteUntracked = AURORA_GUEST_WRITE_UNTRACKED; inline AuroraGuestWriteGenerationCallback g_guestWriteGenerationHook = nullptr; inline AuroraGuestWriteNotifyCallback g_guestWriteNotifyHook = nullptr; inline uint64_t guest_write_generation(const void* data, size_t size) noexcept { if (g_guestWriteGenerationHook == nullptr || data == nullptr || size == 0) { return kGuestWriteUntracked; } return g_guestWriteGenerationHook(data, size); } inline void notify_guest_write(const void* data, size_t size) noexcept { if (g_guestWriteNotifyHook == nullptr || data == nullptr || size == 0) { return; } g_guestWriteNotifyHook(data, size); } // True when `stored` was taken over the same untouched bytes, so its digest still describes them. // The untracked sentinel never matches: a source aurora cannot watch is always re-digested. inline bool guest_write_generation_matches(uint64_t stored, uint64_t current) noexcept { return current != kGuestWriteUntracked && stored == current; } class Hasher { public: explicit Hasher(const XXH64_hash_t seed = 0) { XXH3_INITSTATE(&state); XXH3_64bits_reset_withSeed(&state, seed); } void update(const void* data, const size_t size) { XXH3_64bits_update(&state, data, size); } template void update(const T& data) { static_assert(std::has_unique_object_representations_v); update(&data, sizeof(T)); } [[nodiscard]] XXH64_hash_t digest() const { return XXH3_64bits_digest(&state); } private: XXH3_state_t state; }; class ByteBuffer { public: ByteBuffer() noexcept = default; explicit ByteBuffer(size_t size) noexcept : m_data(static_cast(calloc(1, size))), m_length(size), m_capacity(size) {} explicit ByteBuffer(uint8_t* data, size_t size) noexcept : m_data(data), m_capacity(size), m_owned(false) {} ~ByteBuffer() noexcept { if (m_data != nullptr && m_owned) { free(m_data); } } ByteBuffer(ByteBuffer&& rhs) noexcept : m_data(rhs.m_data), m_length(rhs.m_length), m_capacity(rhs.m_capacity), m_owned(rhs.m_owned) { rhs.m_data = nullptr; rhs.m_length = 0; rhs.m_capacity = 0; rhs.m_owned = true; } ByteBuffer& operator=(ByteBuffer&& rhs) noexcept { if (m_data != nullptr && m_owned) { free(m_data); } m_data = rhs.m_data; m_length = rhs.m_length; m_capacity = rhs.m_capacity; m_owned = rhs.m_owned; rhs.m_data = nullptr; rhs.m_length = 0; rhs.m_capacity = 0; rhs.m_owned = true; return *this; } ByteBuffer(ByteBuffer const&) = delete; ByteBuffer& operator=(ByteBuffer const&) = delete; operator ArrayRef() const noexcept { return {m_data, m_length}; } [[nodiscard]] uint8_t* data() noexcept { return m_data; } [[nodiscard]] const uint8_t* data() const noexcept { return m_data; } [[nodiscard]] size_t size() const noexcept { return m_length; } [[nodiscard]] bool empty() const noexcept { return m_length == 0; } void append(const void* data, size_t size) { resize(m_length + size, false); memcpy(m_data + m_length, data, size); m_length += size; } template void append(const T& obj) { append(&obj, sizeof(T)); } void append_zeroes(size_t size) { resize(m_length + size, true); m_length += size; } // Extend the buffer without clearing the new bytes. Only for callers that overwrite the whole // region: the mapped staging buffers are megabytes of write-combine memory. void append_uninitialized(size_t size) { resize(m_length + size, false); m_length += size; } void release() { if (m_data != nullptr && m_owned) { free(m_data); } m_data = nullptr; m_length = 0; m_capacity = 0; m_owned = true; } void clear() { m_length = 0; } void reserve_extra(size_t size) { resize(m_length + size, true); } ByteBuffer clone() const { ByteBuffer clone{m_length}; std::memcpy(clone.data(), m_data, m_length); return clone; } private: uint8_t* m_data = nullptr; size_t m_length = 0; size_t m_capacity = 0; bool m_owned = true; // `size` is the total capacity needed. When `zeroed` is set, [m_length, size) has to read back as // zero on every branch; the early return used to leave the previous frame's bytes in the padding. void resize(size_t size, bool zeroed) { if (size == 0) { clear(); return; } const size_t zeroBegin = m_length; if (m_data == nullptr) { if (zeroed) { m_data = static_cast(calloc(1, size)); } else { m_data = static_cast(malloc(size)); } m_owned = true; m_capacity = size; // calloc already cleared the whole allocation. return; } if (size > m_capacity) { if (!m_owned) { abort(); } // Exponential expansion to avoid O(n^2) time complexity. size_t capacity = size; if (capacity < m_capacity * 2) { capacity = m_capacity * 2; } m_data = static_cast(realloc(m_data, capacity)); m_capacity = capacity; } if (zeroed && size > zeroBegin) { memset(m_data + zeroBegin, 0, size - zeroBegin); } } }; } // namespace aurora namespace aurora::gfx { inline constexpr bool UseTextureBuffer = false; inline constexpr uint64_t UniformBufferSize = 25165824; // 24mb inline constexpr uint64_t VertexBufferSize = 3145728; // 3mb inline constexpr uint64_t IndexBufferSize = 2097152; // 2mb inline constexpr uint64_t StorageBufferSize = 8388608; // 8mb inline constexpr uint64_t TextureUploadSize = 25165824; // 24mb extern AuroraStats g_stats; extern uint32_t g_drawCallCount; extern uint32_t g_mergedDrawCallCount; extern wgpu::Buffer g_vertexBuffer; extern wgpu::Buffer g_uniformBuffer; extern wgpu::Buffer g_indexBuffer; extern wgpu::Buffer g_storageBuffer; extern wgpu::BindGroupLayout g_staticBindGroupLayout; extern wgpu::BindGroup g_staticBindGroup; extern wgpu::BindGroupLayout g_uniformBindGroupLayout; extern wgpu::BindGroup g_uniformBindGroup; using BindGroupRef = HashType; using PipelineRef = HashType; using SamplerRef = HashType; using ShaderRef = HashType; struct Range { uint32_t offset = 0; uint32_t size = 0; bool operator==(const Range& rhs) const { return memcmp(this, &rhs, sizeof(*this)) == 0; } bool operator!=(const Range& rhs) const { return !(*this == rhs); } }; struct ClipRect { int32_t x; int32_t y; int32_t width; int32_t height; bool operator==(const ClipRect& rhs) const { return memcmp(this, &rhs, sizeof(*this)) == 0; } bool operator!=(const ClipRect& rhs) const { return !(*this == rhs); } }; using webgpu::Viewport; struct TextureRef; using TextureHandle = std::shared_ptr; enum class ShaderType : uint8_t { Clear = 0, GX = 1, }; void initialize(); void shutdown(); bool begin_frame(); bool resume_frame(); void abort_frame() noexcept; struct ReplayTarget { wgpu::TextureView colorView; // A view of the same texture that a fragment density map is bound to (webgpu/fdm.hpp), set when // this frame's eyes are foveated. Only an eye drawn in a single render pass renders through it: a // later render pass would load the eye back under the density map. wgpu::TextureView foveatedColorView; wgpu::TextureView resolveView; wgpu::TextureView depthView; wgpu::Texture copySourceTexture; wgpu::TextureView copySourceView; wgpu::TextureView copySourceDepthView; wgpu::Extent3D size{}; uint32_t msaaSamples = 1; wgpu::TextureFormat depthFormat = wgpu::TextureFormat::Depth32Float; }; struct StereoReplayEye { ReplayTarget target; Mat4x4 projection; // The headset's eye delta, from the VR-neutral view space into this eye's. // The virtual screen is defined in that neutral space, so 2D reprojection // uses this transform directly. Mat3x4 viewFromCenter; // The same delta with the first-person scene anchor folded in, i.e. from the // game's *recorded* view space into this eye's. World draws carry the game's // camera in their position matrices and therefore need this one. It equals // viewFromCenter whenever the anchor is identity. Mat3x4 viewFromScene; // The same transform for a draw whose projection mirrors X (Mario Kart Wii's // mirror mode), built from the mirrored eye delta so the reflection lands in // the anchored camera's space rather than in each eye's own. Pairs with // stereo_replay::mirror_projection_x; see the comment on those helpers. Mat3x4 viewFromSceneMirrored; }; struct StereoReplayFrame { std::array eyes; // VR hands and synthetic wheel, drawn per eye after the world (gfx/cockpit.hpp). AuroraCockpit cockpit{}; }; void end_frame(const wgpu::CommandEncoder& cmd); // Set under the renderer mutex immediately before preparing/sealing the frame. void set_stereo_local_player_count(uint32_t count) noexcept; // Prepares eye-specific uniform copies before unmapping the staging buffer. // Returns false without modifying the mono path when the uniform buffer has // insufficient room for the additional copies. bool end_frame(const wgpu::CommandEncoder& cmd, const StereoReplayFrame& stereoFrame); void end_batch(const wgpu::CommandEncoder& cmd); uint32_t current_frame() noexcept; // One frame's recorded render passes, detached from the guest-visible // recording state. The contents are private to common.cpp. struct SealedFrameData; // Owns the recorded passes of a sealed frame, touched only by the thread that sealed it. // seal_frame() runs with the producer excluded, and the object is reused to keep its capacity. class SealedFrame { public: SealedFrame(); ~SealedFrame(); SealedFrame(SealedFrame&&) noexcept; SealedFrame& operator=(SealedFrame&&) noexcept; SealedFrame(const SealedFrame&) = delete; SealedFrame& operator=(const SealedFrame&) = delete; [[nodiscard]] SealedFrameData& data() const noexcept { return *m_data; } private: std::unique_ptr m_data; }; // Detach the recorded passes of the frame that just ended into `out`. Must be // called with the renderer GPU mutex held; see SealedFrame. void seal_frame(SealedFrame& out) noexcept; // Retained replay owns CPU transform endpoints and reserved eye-uniform ranges. // Call with the renderer mutex held: a mid-frame EFB submission invalidates it. bool has_late_stereo_replay(const SealedFrame& frame) noexcept; bool prepare_late_stereo_replay(SealedFrame& frame, wgpu::CommandEncoder& cmd, const StereoReplayFrame& stereoFrame, float weight); // Encode a sealed frame. Never touches the producer-visible recording state, // so this may run concurrently with the producer's FIFO drains. // `nativeRenderLastPass` limits the passes that do render work (texture bakes still run for every // pass): a headset never shows an immersive frame's native render, so encode_sealed_frame stops it // after the last pass whose EFB copy the eye replays sample. void render(SealedFrame& frame, wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true, int32_t nativeRenderLastPass = INT32_MAX); // Index of the last recorded pass that resolves an EFB copy other than the display copy, or -1 // when no pass does: everything after it exists only for the presented image. int32_t last_pass_feeding_replay(const SealedFrame& frame) noexcept; // Replays only main-EFB passes into one Aurora-owned eye target. Native // offscreen/EFB-copy passes are consumed from the mono render and are not // mutated by stereo replay. void render_stereo_eye(SealedFrame& frame, wgpu::CommandEncoder& cmd, const StereoReplayFrame& stereoFrame, uint32_t eye, bool finalize = false); // Encode the frame that is still being recorded. Only for the synchronous // EFB-readback split path, which runs on the producer thread. void render(wgpu::CommandEncoder& cmd, int32_t interpolatedFrame = -1, bool finalize = true); // Sweep bind groups that have not been used for a while. Mutates the cache the producer inserts // into, so it may only run with the producer excluded, inside the seal prologue. void expire_bind_group_cache() noexcept; void after_submit() noexcept; void map_staging_buffer(); // `persistentCopy` marks resolves whose destination outlives the frame with no re-issue path, so // the pass waits for its pipelines instead of dropping draws that would be baked in permanently. void resolve_pass(TextureHandle texture, ClipRect rect, bool clearColor, bool clearAlpha, bool clearDepth, Vec4 clearColorValue, float clearDepthValue, GXTexFmt resolveFormat = GX_TF_RGBA8, const Vec4* sourceRectPixels = nullptr, bool halfScale = false, const std::array* copyFilterCoefficients = nullptr, bool forceOpaqueAlpha = false, float copyFilterRowStride = 1.0f, bool clampTop = false, bool clampBottom = false, bool persistentCopy = false); // Marks the resolve immediately preceding the current continuation pass as the // EFB-to-display copy. Immersive replay uses its source rectangle as the eye // viewport instead of exposing the Wii's larger scratch EFB workspace. void mark_last_resolve_as_display_copy() noexcept; // Immersive-replay EFB controls, both on by default. Safe to flip at any time: // the frame worker reads them atomically once per eye. void set_stereo_stop_at_display_copy(bool value) noexcept; bool get_stereo_stop_at_display_copy() noexcept; void set_stereo_skip_copy_clears(bool value) noexcept; bool get_stereo_skip_copy_clears() noexcept; // Replays each eye in as few render passes as its clears allow (gfx/eye_pass_plan.hpp). On by // default and live, like the two above. void set_stereo_single_pass_eyes(bool value) noexcept; bool get_stereo_single_pass_eyes() noexcept; // Foveated rendering level for the immersive eyes (gfx/foveation.hpp Level). Live: the next frame's // eyes use it, provided the device has fragment density maps. void set_stereo_foveation(uint32_t level) noexcept; uint32_t get_stereo_foveation() noexcept; // Places orthographic draws on a fixed virtual screen during immersive replay. // `width` and `distance` are in game world units; the screen's height follows // the game's presented aspect ratio. Also live. void set_stereo_hud_screen(bool enabled, float width, float distance) noexcept; bool get_stereo_hud_screen_enabled() noexcept; // The screen's width and distance in world units, kept while the 2D layer is off it. void get_stereo_hud_screen_size(float& width, float& distance) noexcept; // What the desktop window presents while a stereo provider is feeding a // headset. Live, and read once per presentation group by the frame worker. void set_stereo_mirror_view(AuroraStereoMirrorView value) noexcept; AuroraStereoMirrorView get_stereo_mirror_view() noexcept; void begin_offscreen(uint32_t width, uint32_t height); void end_offscreen(); bool is_offscreen() noexcept; uint32_t get_sample_count() noexcept; void clear_caches() noexcept; // Per-pass GPU timing for the frame-rate log. When enabled and the device has TimestampQuery, // every render or compute pass asks gpu_timing_pass() for timestamp writes under a category; the // frame's queries are resolved into a small ring of readback buffers and the completed frames' // durations are summed per category until gpu_timing_report() consumes them. Off by default: // aurora.cpp enables it together with the Android frame-rate log. enum class GpuTimingCategory : uint8_t { Mono, // the native (desktop) render of the recorded GX passes EyeLeft, // stereo replay of the left eye EyeRight, // stereo replay of the right eye Interpolated, // interpolated presentation slots VirtualScreen, // the 2D virtual screen built for each eye Panel, // the in-headset settings panel EfbCopy, // EFB copy format conversions Palette, // palette (TLUT) texture conversions DepthPeek, // the depth snapshot compute pass Snapshot, // presentation snapshot and its ImGui pass Present, // the desktop presentation copy Count, }; void gpu_timing_set_enabled(bool enabled) noexcept; bool gpu_timing_enabled() noexcept; // Opens the current frame's query slot; a frame whose slot is still being read back is skipped. void gpu_timing_begin_frame() noexcept; // Timestamp writes for one pass of the open frame, or nullptr when timing is off or exhausted. const wgpu::PassTimestampWrites* gpu_timing_pass(GpuTimingCategory category) noexcept; // Resolves the open frame's queries on `encoder`, which must be the frame's last submission. void gpu_timing_end_frame(wgpu::CommandEncoder& encoder) noexcept; // After that submission: starts the readback of the resolved queries. void gpu_timing_after_submit() noexcept; // Per-frame averages of the frames read back since the last call, formatted for the log, or // an empty string when nothing was measured. std::string gpu_timing_report(); namespace tex_palette_conv { struct ConvRequest; } // namespace tex_palette_conv void queue_palette_conv(tex_palette_conv::ConvRequest req); Range push_verts(const uint8_t* data, size_t length); template static Range push_verts(ArrayRef data) { return push_verts(reinterpret_cast(data.data()), data.size() * sizeof(T)); } Range push_indices(const uint8_t* data, size_t length); template static Range push_indices(ArrayRef data) { return push_indices(reinterpret_cast(data.data()), data.size() * sizeof(T)); } Range push_uniform(const uint8_t* data, size_t length); template static Range push_uniform(const T& data) { return push_uniform(reinterpret_cast(&data), sizeof(T)); } Range push_storage(const uint8_t* data, size_t length); template static Range push_storage(ArrayRef data) { return push_storage(reinterpret_cast(data.data()), data.size() * sizeof(T)); } template static Range push_storage(const T& data) { return push_storage(reinterpret_cast(&data), sizeof(T)); } Range push_texture_data(const uint8_t* data, size_t length, uint32_t bytesPerRow, uint32_t rowsPerImage); std::pair map_verts(size_t length); std::pair map_indices(size_t length); std::pair map_uniform(size_t length); std::pair copy_uniform(Range source); std::pair map_storage(size_t length); template const State& get_state(); template void push_draw_command(DrawData data); template DrawData* get_last_draw_command(); template PipelineRef pipeline_ref(const PipelineConfig& config); // `currentPipeline` is the caller's per-pass dedupe slot; as a file-static, two concurrent encoders // skipped each other's SetPipeline. `requireReady` refuses the skip-unready shortcut for bakes. bool bind_pipeline(PipelineRef ref, const wgpu::RenderPassEncoder& pass, PipelineRef& currentPipeline, bool requireReady = false); BindGroupRef bind_group_ref(const WGPUBindGroupDescriptor& descriptor); wgpu::BindGroup& find_bind_group(BindGroupRef id); wgpu::Sampler& sampler_ref(const wgpu::SamplerDescriptor& descriptor); uint32_t align_uniform(uint32_t value); Vec2 get_render_target_size() noexcept; // Same value as get_render_target_size() outside a render pass, but never // touches the frame worker's render-pass list, so it is safe off-thread. Vec2 get_frame_buffer_size() noexcept; void set_viewport(const Viewport& viewport) noexcept; void set_scissor(const ClipRect& scissor) noexcept; void push_debug_group(std::string label); void insert_debug_marker(std::string label); } // namespace aurora::gfx