diff --git a/aurora-main/lib/dolphin/gx/GXAurora.cpp b/aurora-main/lib/dolphin/gx/GXAurora.cpp index 304b86b..f2776f1 100644 --- a/aurora-main/lib/dolphin/gx/GXAurora.cpp +++ b/aurora-main/lib/dolphin/gx/GXAurora.cpp @@ -26,6 +26,8 @@ extern "C" void aurora_clear_native_wheel_vertices() { aurora::gx::nativeWheelPreviousSources.push_back(array.source); aurora::gx::native_wheel_report(); aurora::gx::nativeWheelArrays.clear(); + aurora::gx::nativeWheelLastDecision = nullptr; + aurora::gx::nativeWheelLastDrawCommand = nullptr; } extern "C" uint32_t aurora_native_wheel_draw_count() { return aurora::gx::nativeWheelLastMatches.load(); } extern "C" void aurora_set_native_wheel_vertices(const void* source, const void* replacement, uint32_t size, @@ -37,6 +39,10 @@ extern "C" void aurora_set_native_wheel_vertices(const void* source, const void* array.bytes.assign(bytes, bytes + size); std::memcpy(array.modelView.data(), modelView, sizeof(float) * 12); aurora::gx::nativeWheelArrays.push_back(std::move(array)); + // The vector may have moved its elements, and a set changes what a draw resolves to in any case, so the decision + // the merge test compares against is dropped rather than left pointing into the old storage. + aurora::gx::nativeWheelLastDecision = nullptr; + aurora::gx::nativeWheelLastDrawCommand = nullptr; } // Single definition for the `Log` that gx.hpp declares for this directory. diff --git a/aurora-main/lib/gx/command_processor.cpp b/aurora-main/lib/gx/command_processor.cpp index 8bc2b9c..1f1309f 100644 --- a/aurora-main/lib/gx/command_processor.cpp +++ b/aurora-main/lib/gx/command_processor.cpp @@ -1930,7 +1930,8 @@ static u32 calculate_last_vtx_size(GXVtxFmt fmt) { static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange, uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature, - bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride); + bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride, + NativeWheelArray* nativeWheel); // The per-draw geometry signature, matrix-usage mask and draw-identity hashes exist purely to feed frame interpolation // (build_uniform consumes them only after its `frame_interpolation_fps() == 0` early-out). @@ -1960,6 +1961,25 @@ static uint32_t matrix_index_prefix_size(GXVtxFmt fmt) noexcept { return size; } +// Which animated vertex array, if any, this draw takes (native_wheel.hpp). Called once per draw, before the merge +// test, because a merged draw renders through the binding the draw it folds into resolved. +static NativeWheelArray* resolve_native_wheel(GXVtxFmt fmt, const uint8_t* vertices, u16 vtxCount, + uint32_t vtxStride) noexcept { + // A direct-position draw reads no array at all, and g_gxState.arrays[GX_VA_POS] then still holds whatever was + // bound last, which must not be matched against. + if (g_gxState.vtxDesc[GX_VA_POS] != GX_INDEX8 && g_gxState.vtxDesc[GX_VA_POS] != GX_INDEX16) + LIKELY { return nullptr; } + const auto& array = g_gxState.arrays[GX_VA_POS]; + if (nativeWheelArrays.empty()) + LIKELY { + if (!nativeWheelPreviousSources.empty()) + UNLIKELY { native_wheel_note_outside(array.data); } + return nullptr; + } + return native_wheel_array(array, vertices, static_cast(vtxCount) * vtxStride, vtxStride, + matrix_index_prefix_size(fmt)); +} + // Screen-space bounds of a simple orthographic rectangle or line, textured or // not: MKW's split-screen partition is a layout picture pane (a one-pixel quad // sampling a pattern texture), so texture use cannot disqualify a candidate. @@ -2322,7 +2342,8 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{}; handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature, interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0, - interpolationIdentityActive, vertices, vtxSize); + interpolationIdentityActive, vertices, vtxSize, + resolve_native_wheel(fmt, vertices, vtxCount, vtxSize)); return true; } @@ -2355,15 +2376,21 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi gfx::Range vertRange = push_draw_vertices(vertices, vtxCount, vtxSize); pos += totalVtxBytes; - // Try to merge with previous draw call. A draw that binds a native-wheel source array is decided per draw - // (native_wheel_array), so it never folds into a neighbour that resolved the array differently. - if (!g_gxState.stateDirty && !(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC) && - !native_wheel_source(g_gxState.arrays[GX_VA_POS].data)) + // The animated vertex array this draw takes is decided per draw, and the decision is part of what a merge would + // share, so resolve it here and hand the result to handle_draw_unmerged rather than deciding twice. + NativeWheelArray* const nativeWheel = resolve_native_wheel(fmt, vertices, vtxCount, vtxSize); + + // Try to merge with previous draw call. + if (!g_gxState.stateDirty && !(aurora::stereo_frame_provider_active() && g_gxState.projType == GX_ORTHOGRAPHIC)) LIKELY { auto* lastDraw = gfx::get_last_draw_command(); - // Only if the previous draw call was a single instance draw (no lines/points handling) + // Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw + // that resolved the same animated array: the merged whole renders through that draw's binding. Anything the + // decision cache cannot vouch for (a command it was not recorded against) stays unmerged. if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS && - lastDraw->instanceCount == 1) + lastDraw->instanceCount == 1 && + (nativeWheelArrays.empty() || + (nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel))) LIKELY { const auto& indexTemplate = cached_index_template(prim, vtxCount); const auto indices = offset_index_template(indexTemplate, lastDraw->vtxCount); @@ -2392,13 +2419,14 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi const PnMtxUsage matrixUsage = interpolationIdentityActive ? pn_mtx_usage(vertices, vtxCount, vtxSize) : PnMtxUsage{}; handle_draw_unmerged(prim, fmt, vtxCount, vertRange, matrixUsage.mask, matrixUsage.topologySignature, interpolationIdentityActive ? draw_geometry_signature(fmt, vertices, vtxCount, vtxSize) : 0, - interpolationIdentityActive, vertices, vtxSize); + interpolationIdentityActive, vertices, vtxSize, nativeWheel); return true; } static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, gfx::Range vertRange, uint16_t usedPnMtxMask, HashType matrixTopologySignature, HashType geometrySignature, - bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride) { + bool interpolationIdentityActive, const uint8_t* vertices, uint32_t vtxStride, + NativeWheelArray* nativeWheel) { ZoneScoped; // GX_CULL_ALL rasterizes nothing on hardware - no color, no depth. if (g_gxState.cullMode == GX_CULL_ALL && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS) @@ -2422,28 +2450,23 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g } auto& array = g_gxState.arrays[i]; const u32 uploadStride = padded_upload_stride(array.stride); - if (i == GX_VA_POS && nativeWheelArrays.empty() && !nativeWheelPreviousSources.empty()) - UNLIKELY { native_wheel_note_outside(array.data); } - if (i == GX_VA_POS && !nativeWheelArrays.empty()) + if (i == GX_VA_POS && nativeWheel != nullptr) UNLIKELY { - if (auto* nativeWheel = native_wheel_array(array, vertices, static_cast(vtxCount) * vtxStride, vtxStride, - matrix_index_prefix_size(fmt))) { - static unsigned nativeWheelDrawLogs = 0; - if (nativeWheelDrawLogs++ < 4) Log.info("Native steering wheel: animated local vehicle vertex array"); - // Never populate the shared source's cache with the animated copy: later draws of the same asset must - // still see the original vertices. The copy takes the same padded upload path as the original. - if (nativeWheel->uploaded.size == 0 || nativeWheel->uploadedStride != uploadStride) { - AttrArray animated{}; - animated.data = nativeWheel->bytes.data(); - animated.size = array.size; - animated.stride = array.stride; - animated.le = array.le; - nativeWheel->uploaded = push_vertex_array(animated, uploadStride); - nativeWheel->uploadedStride = uploadStride; - } - ranges.vaRanges[0] = nativeWheel->uploaded; - continue; + static unsigned nativeWheelDrawLogs = 0; + if (nativeWheelDrawLogs++ < 4) Log.info("Native steering wheel: animated local vehicle vertex array"); + // Never populate the shared source's cache with the animated copy: later draws of the same asset must + // still see the original vertices. The copy takes the same padded upload path as the original. + if (nativeWheel->uploaded.size == 0 || nativeWheel->uploadedStride != uploadStride) { + AttrArray animated{}; + animated.data = nativeWheel->bytes.data(); + animated.size = array.size; + animated.stride = array.stride; + animated.le = array.le; + nativeWheel->uploaded = push_vertex_array(animated, uploadStride); + nativeWheel->uploadedStride = uploadStride; } + ranges.vaRanges[0] = nativeWheel->uploaded; + continue; } if (array.cachedRange.size > 0 && array.cachedStride == uploadStride) { ranges.vaRanges[i - GX_VA_POS] = array.cachedRange; @@ -2516,6 +2539,9 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g .dstAlpha = pipelineState.dstAlpha, .screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride), }); + // What the next draw must match to be allowed to fold into this one. + nativeWheelLastDrawCommand = gfx::get_last_draw_command(); + nativeWheelLastDecision = nativeWheel; g_gxState.stateDirty = false; } diff --git a/aurora-main/lib/gx/native_wheel.hpp b/aurora-main/lib/gx/native_wheel.hpp index 5bcbd48..bd80e6a 100644 --- a/aurora-main/lib/gx/native_wheel.hpp +++ b/aurora-main/lib/gx/native_wheel.hpp @@ -28,6 +28,16 @@ inline std::vector nativeWheelArrays; inline uint32_t nativeWheelMatches=0; inline std::atomic nativeWheelLastMatches{0}; +// A draw folds into the previous one by appending its vertices to that draw's +// range, so the merged whole renders through the *first* draw's array binding: +// two draws may only merge when they resolved the same replacement. The +// decision below is therefore taken once per draw (the ownership walk is far +// too costly to repeat) and kept with the command it was recorded for. It is +// cleared with the set, which the runtime posts once a frame, so a command +// address a later frame's list reuses can never be read as a hit. +inline NativeWheelArray* nativeWheelLastDecision=nullptr; +inline const void* nativeWheelLastDrawCommand=nullptr; + // For the host log: why draws of the replaced arrays did or did not take them. struct NativeWheelDiagnostics { uint32_t sets=0; // replacement sets cleared since the last report diff --git a/aurora-main/tests/native_wheel_test.cpp b/aurora-main/tests/native_wheel_test.cpp index 2ef0407..5f46553 100644 --- a/aurora-main/tests/native_wheel_test.cpp +++ b/aurora-main/tests/native_wheel_test.cpp @@ -4,7 +4,9 @@ #include #include +#include +#include "gx_test_common.hpp" #include "gx/native_wheel.hpp" namespace { @@ -133,4 +135,73 @@ TEST_F(NativeWheelArrayTest, NonFiniteMatrixNeverMatches) { EXPECT_EQ(aurora::gx::native_wheel_array(array, nullptr, 0, 0, 0), nullptr); } +// Draw merging around an animated array. A merged draw appends its vertices to the previous draw's range and renders +// through that draw's array binding, so two draws may merge only when they resolved the same replacement. The kart's +// display list is hundreds of same-state primitives, and merging them is worth several ms an eye on a tiler. +class NativeWheelMergeTest : public GXFifoTest { +protected: + void SetUp() override { + GXFifoTest::SetUp(); + aurora::gfx::testing::use_draw_command_tracking(true); + aurora::gx::nativeWheelArrays.clear(); + aurora::gx::nativeWheelLastDecision = nullptr; + aurora::gx::nativeWheelLastDrawCommand = nullptr; + aurora::gx::NativeWheelArray replacement; + replacement.source = source.data(); + replacement.bytes.assign(source.size(), 0); + replacement.bytes[12] = 1; // one animated position + aurora::gx::nativeWheelArrays.push_back(replacement); + auto& state = aurora::gx::g_gxState; + state.lastVtxFmt = GX_VTXFMT0; + state.lastVtxSize = 1; + state.vtxDesc[GX_VA_POS] = GX_INDEX8; + state.arrays[GX_VA_POS].data = source.data(); + state.arrays[GX_VA_POS].size = static_cast(source.size()); + state.arrays[GX_VA_POS].stride = 12; + state.stateDirty = true; + } + void TearDown() override { + aurora::gx::nativeWheelArrays.clear(); + aurora::gx::nativeWheelLastDecision = nullptr; + aurora::gx::nativeWheelLastDrawCommand = nullptr; + } + void draw() { + std::vector fifo{static_cast(GX_TRIANGLES) | static_cast(GX_VTXFMT0), 0, 3, 0, 1, 2}; + decode_fifo(fifo); + } + // The local vehicle's matrix: the replacement's model-view is all zeroes, and so is a default palette slot. + void makeOpponent() { reinterpret_cast(&aurora::gx::g_gxState.pnMtx[0].pos)[3] = 100.f; } + std::array source{}; +}; + +TEST_F(NativeWheelMergeTest, PrimitivesSharingTheAnimatedArrayStillMerge) { + draw(); + ASSERT_EQ(aurora::gx::nativeWheelLastDecision, &aurora::gx::nativeWheelArrays.front()); + draw(); + EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u); +} + +TEST_F(NativeWheelMergeTest, PrimitivesThatTakeTheOriginalArrayAlsoStillMerge) { + makeOpponent(); + draw(); + ASSERT_EQ(aurora::gx::nativeWheelLastDecision, nullptr); + draw(); + EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 1u); +} + +TEST_F(NativeWheelMergeTest, OpponentPrimitiveNeverFoldsIntoAnAnimatedDraw) { + draw(); + makeOpponent(); + draw(); + EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u) << "the merged whole would render the opponent animated"; +} + +TEST_F(NativeWheelMergeTest, AnimatedPrimitiveNeverFoldsIntoAnOpponentDraw) { + makeOpponent(); + draw(); + reinterpret_cast(&aurora::gx::g_gxState.pnMtx[0].pos)[3] = 0.f; + draw(); + EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u) << "the wheel would render on the original vertices"; +} + } // namespace