Stop draws from merging or reading past their own vertices

Incomplete primitives no longer emit indices: quads drop a one- or
two-vertex tail and draw a three-vertex tail as a triangle, triangle
lists ignore leftover vertices, and fans or strips under three vertices
emit nothing. Draws with no complete primitive are skipped. Previously an
incomplete quad indexed vertices that belonged to the next merged draw,
or past the end of the buffer.

Merging now stops before the 16-bit index offset would wrap, never folds
triangles into a single-instance line or point draw, and breaks after
GXInvalidateVtxCache or a vertex-format switch, so the next draw uploads
fresh arrays and uses its own format's shader.

From upstream patchzyy/Wiicompiled 6f14bde (#244, KartPad batch).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Wg7mB8ogCWmp9GH19Uc82B
This commit is contained in:
Claude committed 2026-10-05 13:29:17 +00:00
1 parent c12b6b7d99
commit b361ecd54c
5 files changed
+130 -13

No files matched your search

+40 -12
View File
@@ -33,10 +33,11 @@ using IndexBuffer = std::vector<u16>;
static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount) { static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount) {
size_t writePos = 0; size_t writePos = 0;
if (prim == GX_QUADS) { if (prim == GX_QUADS) {
// Retain the existing incomplete-quad behavior: every started group emits a complete six-index quad. // GX renders a three-vertex remainder as a triangle. One/two are ignored.
buf.resize(((static_cast<u32>(vtxCount) + 3u) / 4u) * 6u); const u32 completeVertices = static_cast<u32>(vtxCount) & ~3u;
buf.resize((completeVertices / 4u) * 6u + (vtxCount % 4u == 3u ? 3u : 0u));
for (u16 v = 0; v < vtxCount; v += 4) { for (u32 v = 0; v < completeVertices; v += 4) {
const u16 idx0 = v; const u16 idx0 = v;
const u16 idx1 = static_cast<u16>(v + 1); const u16 idx1 = static_cast<u16>(v + 1);
const u16 idx2 = static_cast<u16>(v + 2); const u16 idx2 = static_cast<u16>(v + 2);
@@ -48,15 +49,21 @@ static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount
buf[writePos++] = idx3; buf[writePos++] = idx3;
buf[writePos++] = idx0; buf[writePos++] = idx0;
} }
if (vtxCount % 4u == 3u) {
buf[writePos++] = static_cast<u16>(completeVertices);
buf[writePos++] = static_cast<u16>(completeVertices + 1u);
buf[writePos++] = static_cast<u16>(completeVertices + 2u);
}
} else if (prim == GX_TRIANGLES) { } else if (prim == GX_TRIANGLES) {
buf.resize(vtxCount); const u32 completeVertices = (static_cast<u32>(vtxCount) / 3u) * 3u;
for (u16 v = 0; v < vtxCount; ++v) { buf.resize(completeVertices);
for (u32 v = 0; v < completeVertices; ++v) {
buf[writePos++] = v; buf[writePos++] = v;
} }
} else if (prim == GX_TRIANGLEFAN) { } else if (prim == GX_TRIANGLEFAN) {
const u32 indexCount = vtxCount <= 3 ? vtxCount : 3u + (static_cast<u32>(vtxCount) - 3u) * 3u; const u32 indexCount = vtxCount < 3 ? 0u : (static_cast<u32>(vtxCount) - 2u) * 3u;
buf.resize(indexCount); buf.resize(indexCount);
for (u16 v = 0; v < vtxCount; ++v) { for (u32 v = 0; indexCount != 0 && v < vtxCount; ++v) {
if (v < 3) { if (v < 3) {
buf[writePos++] = v; buf[writePos++] = v;
continue; continue;
@@ -66,9 +73,9 @@ static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount
buf[writePos++] = v; buf[writePos++] = v;
} }
} else if (prim == GX_TRIANGLESTRIP) { } else if (prim == GX_TRIANGLESTRIP) {
const u32 indexCount = vtxCount <= 3 ? vtxCount : 3u + (static_cast<u32>(vtxCount) - 3u) * 3u; const u32 indexCount = vtxCount < 3 ? 0u : (static_cast<u32>(vtxCount) - 2u) * 3u;
buf.resize(indexCount); buf.resize(indexCount);
for (u16 v = 0; v < vtxCount; ++v) { for (u32 v = 0; indexCount != 0 && v < vtxCount; ++v) {
if (v < 3) { if (v < 3) {
buf[writePos++] = v; buf[writePos++] = v;
continue; continue;
@@ -91,6 +98,13 @@ static u32 prepare_idx_template(IndexBuffer& buf, GXPrimitive prim, u16 vtxCount
return static_cast<u32>(writePos); return static_cast<u32>(writePos);
} }
// Empty/incomplete draws consume FIFO bytes but cannot produce a primitive.
static bool has_complete_primitive(GXPrimitive prim, u16 count) {
if (prim == GX_POINTS) return count >= 1;
if (prim == GX_LINES || prim == GX_LINESTRIP) return count >= 2;
return count >= 3;
}
// GX FIFO opcodes - use CP_ prefix to avoid clashing with GXCommandList.h macros // GX FIFO opcodes - use CP_ prefix to avoid clashing with GXCommandList.h macros
static constexpr u8 CP_CMD_NOP = GX_NOP; static constexpr u8 CP_CMD_NOP = GX_NOP;
static constexpr u8 CP_CMD_LOAD_CP_REG = GX_LOAD_CP_REG; static constexpr u8 CP_CMD_LOAD_CP_REG = GX_LOAD_CP_REG;
@@ -552,6 +566,10 @@ void process(const u8* data, u32 size, bool bigEndian) {
for (int i = GX_VA_POS; i <= GX_VA_TEX7; ++i) { for (int i = GX_VA_POS; i <= GX_VA_TEX7; ++i) {
g_gxState.arrays[i].cachedRange = {}; g_gxState.arrays[i].cachedRange = {};
} }
// A merged draw retains its previous array uploads. Force a new draw so
// handle_draw_unmerged observes the invalidation and uploads fresh data.
// Pipeline configuration itself did not change.
g_gxState.stateDirty = true;
break; break;
} }
@@ -1926,6 +1944,10 @@ static u32 calculate_last_vtx_size(GXVtxFmt fmt) {
g_gxState.lastVtxFmt = fmt; g_gxState.lastVtxFmt = fmt;
g_gxState.lastVtxSize = vtxSize; g_gxState.lastVtxSize = vtxSize;
// The format is selected by the draw opcode, without a register write.
// Even equal-stride formats may decode bytes differently, so do not merge
// into a draw using the previous format's shader and uniform layout.
g_gxState.stateDirty = true;
return vtxSize; return vtxSize;
} }
@@ -2366,6 +2388,8 @@ bool submit_raw_draw(GXPrimitive prim, GXVtxFmt fmt, const uint8_t* vertices, ui
return false; return false;
} }
if (!has_complete_primitive(prim, vtxCount)) return true;
// This entry point bypasses process(), so it owns the renderer lock itself. // This entry point bypasses process(), so it owns the renderer lock itself.
std::lock_guard gpuLock(aurora::renderer_gpu_mutex()); std::lock_guard gpuLock(aurora::renderer_gpu_mutex());
if (model_array_hidden()) return true; if (model_array_hidden()) return true;
@@ -2414,7 +2438,7 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
return false; return false;
} }
if (model_array_hidden()) { if (!has_complete_primitive(prim, vtxCount) || model_array_hidden()) {
pos += totalVtxBytes; pos += totalVtxBytes;
return true; return true;
} }
@@ -2436,9 +2460,12 @@ static bool handle_draw(u8 cmd, const u8* data, u32& pos, u32 size, bool bigEndi
// Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw // Only if the previous draw call was a single instance draw (no lines/points handling), and only into a draw
// that resolved the same animated array: the merged whole renders through that draw's binding. Anything the // that resolved the same animated array: the merged whole renders through that draw's binding. Anything the
// decision cache cannot vouch for (a command it was not recorded against) stays unmerged. // decision cache cannot vouch for (a command it was not recorded against) stays unmerged.
// Expanded lines/points have different vertex interpretation even with one instance.
// Triangle-list output has no restart index; index 65535 is usable.
// Overflow would address earlier vertices instead of the appended geometry.
if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS && if (lastDraw != nullptr && prim != GX_LINES && prim != GX_LINESTRIP && prim != GX_POINTS &&
!lastDraw->uniformReplayLayout.vertexMotion.enabled && !lastDraw->uniformReplayLayout.vertexMotion.enabled && !lastDraw->expandedPrimitive &&
lastDraw->instanceCount == 1 && lastDraw->instanceCount == 1 && uint64_t(lastDraw->vtxCount) + vtxCount <= 65536u &&
(nativeWheelArrays.empty() || (nativeWheelArrays.empty() ||
(nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel))) (nativeWheelLastDrawCommand == lastDraw && nativeWheelLastDecision == nativeWheel)))
LIKELY { LIKELY {
@@ -2615,6 +2642,7 @@ static void handle_draw_unmerged(GXPrimitive prim, GXVtxFmt fmt, u16 vtxCount, g
.vtxCount = vtxCount, .vtxCount = vtxCount,
.indexCount = numIndices, .indexCount = numIndices,
.instanceCount = instanceCount, .instanceCount = instanceCount,
.expandedPrimitive = prim == GX_LINES || prim == GX_LINESTRIP || prim == GX_POINTS,
.bindGroups = bindGroups, .bindGroups = bindGroups,
.dstAlpha = pipelineState.dstAlpha, .dstAlpha = pipelineState.dstAlpha,
.screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride), .screenRect = screen_rect(prim, fmt, vertices, vtxCount, vtxStride),
+1
View File
@@ -29,6 +29,7 @@ struct DrawData {
uint32_t vtxCount; uint32_t vtxCount;
uint32_t indexCount; uint32_t indexCount;
uint32_t instanceCount; uint32_t instanceCount;
bool expandedPrimitive;
GXBindGroups bindGroups; GXBindGroups bindGroups;
uint32_t dstAlpha; uint32_t dstAlpha;
// Valid only for simple orthographic rectangles/lines (textured or not). // Valid only for simple orthographic rectangles/lines (textured or not).
+1
View File
@@ -64,6 +64,7 @@ if (AURORA_ENABLE_GX)
add_executable(gx_fifo_tests add_executable(gx_fifo_tests
gx_fifo_test.cpp gx_fifo_test.cpp
gx_test_stubs.cpp gx_test_stubs.cpp
renderer_regression_test.cpp
stereo_replay_test.cpp stereo_replay_test.cpp
stereo_interpolation_test.cpp stereo_interpolation_test.cpp
stereo_mirror_test.cpp stereo_mirror_test.cpp
+2 -1
View File
@@ -2861,6 +2861,7 @@ TEST_F(GXFifoTest, DrawTopologyTemplatesPreserveExactGxIndexOrder) {
const auto decodeAndReadIndices = [&](GXPrimitive primitive, u16 count) { const auto decodeAndReadIndices = [&](GXPrimitive primitive, u16 count) {
std::vector<u8> fifo; std::vector<u8> fifo;
append_test_draw(fifo, primitive, count); append_test_draw(fifo, primitive, count);
aurora::gfx::testing::reset_vertex_push_record();
decode_fifo(fifo); decode_fifo(fifo);
return aurora::gfx::testing::last_pushed_indices(); return aurora::gfx::testing::last_pushed_indices();
}; };
@@ -2871,7 +2872,7 @@ TEST_F(GXFifoTest, DrawTopologyTemplatesPreserveExactGxIndexOrder) {
g_gxState.stateDirty = true; g_gxState.stateDirty = true;
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 5), (std::vector<u16>{0, 1, 2, 0, 2, 3, 0, 3, 4})); EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 5), (std::vector<u16>{0, 1, 2, 0, 2, 3, 0, 3, 4}));
g_gxState.stateDirty = true; g_gxState.stateDirty = true;
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 2), (std::vector<u16>{0, 1})); EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLEFAN, 2), (std::vector<u16>{}));
g_gxState.stateDirty = true; g_gxState.stateDirty = true;
EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLESTRIP, 6), (std::vector<u16>{0, 1, 2, 2, 1, 3, 2, 3, 4, 4, 3, 5})); EXPECT_EQ(decodeAndReadIndices(GX_TRIANGLESTRIP, 6), (std::vector<u16>{0, 1, 2, 2, 1, 3, 2, 3, 4, 4, 3, 5}));
g_gxState.stateDirty = true; g_gxState.stateDirty = true;
@@ -0,0 +1,86 @@
#include "gx_test_common.hpp"
#include "gx/pipeline.hpp"
using aurora::gx::g_gxState;
namespace {
std::vector<u8> draw(GXPrimitive primitive, u16 count, GXVtxFmt format = GX_VTXFMT0) {
std::vector<u8> bytes{static_cast<u8>(primitive | format), static_cast<u8>(count >> 8), static_cast<u8>(count)};
bytes.resize(3 + count);
return bytes;
}
} // namespace
TEST_F(GXFifoTest, MaximumQuadCountTerminatesWithoutOutOfRangeIndices) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
for (const u16 count : {65532, 65533, 65534, 65535}) {
g_gxState.stateDirty = true;
decode_fifo(draw(GX_QUADS, count));
const auto& indices = aurora::gfx::testing::last_pushed_indices();
ASSERT_EQ(indices.size(), (count / 4) * 6 + (count % 4 == 3 ? 3 : 0));
for (const auto index : indices)
ASSERT_LT(index, count);
}
}
TEST_F(GXFifoTest, IncompletePrimitivesNeverJoinAcrossDraws) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_TRIANGLES, 4));
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
decode_fifo(draw(GX_TRIANGLES, 5));
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{4, 5, 6}));
const auto before = aurora::gfx::testing::last_pushed_indices();
decode_fifo(draw(GX_TRIANGLEFAN, 2));
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), before);
}
TEST_F(GXFifoTest, MergeStopsBeforeSixteenBitIndexOverflow) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_TRIANGLES, 65535));
decode_fifo(draw(GX_TRIANGLES, 3));
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
}
TEST_F(GXFifoTest, VertexCacheInvalidationBreaksDrawMerging) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_TRIANGLES, 3));
decode_fifo({GX_CMD_INVL_VC});
EXPECT_TRUE(g_gxState.stateDirty);
decode_fifo(draw(GX_TRIANGLES, 3));
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
}
TEST_F(GXFifoTest, EqualStrideVertexFormatChangeBreaksDrawMerging) {
aurora::gfx::testing::use_real_vertex_format_helpers(true);
g_gxState.vtxDesc[GX_VA_POS] = GX_DIRECT;
for (const auto format : {GX_VTXFMT0, GX_VTXFMT1}) {
g_gxState.vtxFmts[format].attrs[GX_VA_POS].cnt = GX_POS_XY;
g_gxState.vtxFmts[format].attrs[GX_VA_POS].type = GX_U8;
}
g_gxState.vtxFmts[GX_VTXFMT1].attrs[GX_VA_POS].frac = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
for (const auto format : {GX_VTXFMT0, GX_VTXFMT1}) {
auto bytes = draw(GX_TRIANGLES, 3, format);
bytes.resize(9);
decode_fifo(bytes);
}
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
}
TEST_F(GXFifoTest, SingleExpandedPrimitiveCannotMergeWithTriangles) {
g_gxState.lastVtxFmt = GX_VTXFMT0;
g_gxState.lastVtxSize = 1;
aurora::gfx::testing::use_draw_command_tracking(true);
decode_fifo(draw(GX_POINTS, 1));
decode_fifo(draw(GX_TRIANGLES, 3));
EXPECT_EQ(aurora::gfx::g_mergedDrawCallCount, 0u);
EXPECT_EQ(aurora::gfx::testing::last_pushed_indices(), (std::vector<u16>{0, 1, 2}));
}