Implement GX Thread for Asynchronous Rendering

- Introduced a new GX thread to handle the rendering pipeline, allowing the game thread to post commands without blocking.
- Added `gx_thread.h` and `gx_thread.cpp` to manage the command ring buffer and thread synchronization.
- Updated `vi.cpp` to utilize the GX thread for rendering tasks, improving frame pacing and responsiveness.
- Modified `main.cpp` to configure and start the GX thread, ensuring it integrates with the existing rendering workflow.
- Enhanced `openxr_integration.cpp` to register the GX thread with OpenXR for better performance in VR scenarios.
- Refactored `settings_overlay.cpp` to remove unnecessary waits for the frame worker, as the overlay now draws directly into the game thread's ImGui frame.
- Improved error handling and logging in the GX thread to capture exceptions during command execution.
This commit is contained in:
iChris4 committed 2026-09-19 22:57:49 +02:00
1 parent ffa63fa60a
commit 6a1641e0b7
33 files changed
+1532 -527

No files matched your search

@@ -78,7 +78,7 @@ object GameStorage {
// Written line by line: trimIndent runs after interpolation, so an interpolated line
// would take the indent off every other one.
val lines = mutableListOf(
"# WiiCompiled Quest configuration. Edit with the launcher, the in-game panel or adb pull/push.",
"# WiiCompiled Quest configuration. Edit with the launcher or the in-game panel. After an adb push, run chmod 664 on it or the app can no longer save settings.",
"[paths]",
"dvd_root = \"${discDirectory(context).absolutePath}\"",
)
@@ -227,6 +227,11 @@ class SettingsPage(
read = { it.bool("video", "skip_unready_pipelines") ?: true },
write = { c, value -> c.setBool("video", "skip_unready_pipelines", value) },
)
toggle(
R.string.graphics_gx_thread, R.string.graphics_gx_thread_helper,
read = { it.bool("video", "gx_thread") ?: true },
write = { c, value -> c.setBool("video", "gx_thread", value) },
)
}
}
@@ -188,6 +188,8 @@
<string name="graphics_bloom_helper">Bloom\'s bright glow reads poorly in a headset, so it starts off.</string>
<string name="graphics_skip_unready">Prevent shader stutters</string>
<string name="graphics_skip_unready_helper">Skips a draw for a moment while its shader compiles instead of pausing the game.</string>
<string name="graphics_gx_thread">Graphics thread</string>
<string name="graphics_gx_thread_helper">Prepares the drawing on a second CPU core so busy scenes keep their speed. Turn off only to compare against the single-threaded path. Takes effect on the next launch.</string>
<!-- Controls tab -->
<string name="section_controls">Controllers</string>
+11
View File
@@ -218,6 +218,17 @@ void aurora_end_frame();
// Seal the current frame with an opaque application safety tag. Aurora rejects
// an immersive provider packet unless its contentTag matches this exact frame.
void aurora_end_frame_tagged(uint64_t contentTag);
// aurora_end_frame_tagged() plus the host-owned ImGui frame to present with it (the handle from
// aurora_imgui_host_frame_end(), which this call consumes; NULL presents no host ImGui frame).
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame);
// When the host pumps SDL events itself (aurora_update() on the window's thread) and drives
// begin/end frame from another thread, this stops those calls from pumping events.
void aurora_set_host_event_pump(bool hostPumps);
typedef void (*AuroraFrameLogCallback)(char* buffer, uint32_t bufferSize, double windowSeconds,
uint32_t frames);
// Called with each five-second frame-rate log window (where that log is enabled); a non-empty
// buffer is logged as one extra line.
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback);
/**
* Relocates the immersive camera for the frame about to be sealed.
*
+8
View File
@@ -25,6 +25,14 @@ ImTextureID aurora_imgui_add_texture(uint32_t width, uint32_t height, const void
// producer before the frame is sealed, and leave the draw data untouched until the frame worker is
// done with that frame (aurora_wait_for_frame_worker). Null hides the panel.
void aurora_imgui_set_stereo_overlay(ImDrawData* drawData, float widthFraction);
// Host-owned ImGui frames for the desktop overlay. Begin starts the next frame on the calling
// thread (which must be the window's thread, since the SDL backend reads the window there); end
// renders it and returns a handle to a private copy of its draw data, which aurora_end_frame_ex()
// consumes. A handle that is never presented is freed with aurora_imgui_host_frame_release().
// From the first begin on, aurora no longer starts ImGui frames itself.
void aurora_imgui_host_frame_begin(void);
void* aurora_imgui_host_frame_end(void);
void aurora_imgui_host_frame_release(void* imguiFrame);
#ifdef __cplusplus
}
+84 -23
View File
@@ -69,6 +69,9 @@ AuroraConfig g_config;
uint32_t g_sdlCustomEventsStart;
char g_gameName[4];
std::atomic<AuroraFrameWorkerWaitCallback> g_frameWorkerWaitCallback{nullptr};
// aurora_set_host_event_pump(): the host pumps SDL itself, from the window's thread.
std::atomic_bool g_hostEventPump{false};
std::atomic<AuroraFrameLogCallback> g_frameLogCallback{nullptr};
// Presentation schedule for the frame being sealed, set by the producer. Jobs carry absolute
// deadlines derived from it, so the presenter cannot drift. Zero means present when ready.
std::atomic<uint64_t> g_presentScheduleBaseNanos{0};
@@ -266,7 +269,8 @@ enum class ImGuiFramePolicy {
bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy = ImGuiFramePolicy::Immediate,
bool* imguiNewFrameOwed = nullptr) noexcept;
bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewFrameOwed) noexcept;
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor) noexcept;
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor,
imgui::HostFramePtr hostImGuiFrame) noexcept;
// The two publication points of a frame-worker cycle, cleared together under `mutex`. Sealed:
// producer-shared renderer state is free again. Done: slots encoded, presented, ImGui restarted.
@@ -288,6 +292,7 @@ struct FrameWorkerState {
// belong to that exact queued frame, not to the producer's next frame.
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
StereoSceneAnchor sceneAnchor{};
imgui::HostFramePtr hostImGuiFrame;
// Readiness is polled thousands of times per frame, so these flags double as a publication
// barrier. `sealed` is released before `ready`, and both are cleared under `mutex`.
std::atomic_bool sealed{true};
@@ -338,7 +343,7 @@ bool frame_worker_requested() noexcept {
#ifdef AURORA_ENABLE_GX
// Returns false when a stop request was observed mid-cycle.
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
const StereoSceneAnchor& sceneAnchor) noexcept;
void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept;
#endif
@@ -361,6 +366,7 @@ void frame_worker_main() noexcept {
for (;;) {
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
StereoSceneAnchor sceneAnchor{};
imgui::HostFramePtr hostImGuiFrame;
bool stereoOnly = false;
{
std::unique_lock lock(g_frameWorker.mutex);
@@ -377,6 +383,7 @@ void frame_worker_main() noexcept {
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
sceneAnchor = g_frameWorker.sceneAnchor;
g_frameWorker.sceneAnchor = {};
hostImGuiFrame = std::move(g_frameWorker.hostImGuiFrame);
g_frameWorker.jobPending = false;
}
@@ -396,7 +403,7 @@ void frame_worker_main() noexcept {
g_frameWorker.cv.notify_all();
continue;
}
if (!run_frame_worker_cycle(sealedFrame, contentTag, sceneAnchor)) {
if (!run_frame_worker_cycle(sealedFrame, contentTag, std::move(hostImGuiFrame), sceneAnchor)) {
break;
}
#else
@@ -1556,7 +1563,7 @@ void publish_stereo_screen_aspects(const webgpu::PresentSource& presentSource, c
// gfx::begin_frame() may already have cleared the display-copy override.
void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const webgpu::PresentSource& presentSource,
const PresentationImage& image, bool includeImGui,
MirrorPlan plan = MirrorPlan::Mono) {
MirrorPlan plan = MirrorPlan::Mono, const ImDrawData* hostImGuiData = nullptr) {
ZoneScoped;
auto viewport = webgpu::calculate_present_viewport(image.texture.size.width, image.texture.size.height,
presentSource.size.width, presentSource.size.height);
@@ -1635,7 +1642,11 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
static_cast<float>(image.texture.size.height), 0.f, 1.f);
imgui::render(pass);
if (hostImGuiData != nullptr) {
imgui::render(pass, hostImGuiData);
} else {
imgui::render(pass);
}
pass.End();
}
}
@@ -1678,7 +1689,7 @@ bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy, bool* imgui
ZoneScoped;
#ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost();
if (pumpEvents) {
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
window::pump_events();
}
const bool surfaceReconfigurePending = g_surfaceReconfigurePending.load(std::memory_order_acquire);
@@ -1735,7 +1746,9 @@ bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewF
std::lock_guard gpuLock(g_rendererGpuMutex);
// Note the debt before gfx::begin_frame() can fail: the synchronous path always started the
// ImGui frame here, and the runtime's retry loop depends on that pairing.
if (imguiPolicy == ImGuiFramePolicy::Immediate) {
if (imgui::host_frames_active()) {
// The host starts its own ImGui frames (imgui::host_frame_begin).
} else if (imguiPolicy == ImGuiFramePolicy::Immediate) {
imgui::new_frame(window::get_window_size());
} else if (imguiNewFrameOwed != nullptr) {
*imguiNewFrameOwed = true;
@@ -1771,8 +1784,13 @@ struct SealedFrameContext {
std::optional<AuroraStereoFrame> stereoInput;
bool retainStereo = false;
imgui::StereoOverlay stereoOverlay;
imgui::HostFramePtr imguiFrame;
};
const ImDrawData* host_imgui_data(const SealedFrameContext& ctx) noexcept {
return ctx.imguiFrame ? imgui::host_frame_draw_data(*ctx.imguiFrame) : nullptr;
}
// Worker-owned scene state. A separate buffer generation check protects against
// synchronous EFB submissions overwriting the retained frame's GPU data.
struct RetainedStereoContext {
@@ -1831,7 +1849,7 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
// Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and
// a FIFO already drained into the recorded pass list.
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) {
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) {
ZoneScopedN("Seal frame");
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
// one frame; encode_sealed_frame resolves it on its last submission.
@@ -1889,7 +1907,12 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u
ctx.presentSource = webgpu::current_present_source();
// ImGui draw lists are built once per frame and replayed by each slot's ImGui pass, which is why
// the next ImGui frame cannot start until the encode phase is done.
imgui::render_frame_data();
if (hostImGuiFrame) {
// The host closed its own ImGui frame and handed over a copy of the draw data.
ctx.imguiFrame = std::move(hostImGuiFrame);
} else {
imgui::render_frame_data();
}
// The headset panel's draw data follows the same rule on the host's side.
ctx.stereoOverlay = imgui::latch_stereo_overlay();
// Drop the sealed frame's lazy RAM-readback requests while the producer is still excluded; it
@@ -1981,7 +2004,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false);
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
presentationJobs.push_back({
.image = std::move(image),
.logicalFrame = ctx.logicalFrame,
@@ -2013,7 +2036,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
if (!ctx.replayInterpolatedFrames) {
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
presentationJobs.push_back({
.image = std::move(image),
.logicalFrame = ctx.logicalFrame,
@@ -2053,7 +2076,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
// showing. Black re-clears it below, once the eyes have taken their copy.
const bool virtualScreenNeedsMono = stereoOutput && !immersiveReplay;
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true,
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan);
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan, host_imgui_data(ctx));
if (stereoOutput) {
publish_stereo_screen_aspects(ctx.presentSource, finalImage->texture.size, immersiveReplay);
}
@@ -2075,7 +2098,8 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
stereo_overlay::composite_flat(encoder, output.view, output.size, eye);
}
if (mirrorPlan == MirrorPlan::Black && !headsetOnly) {
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black);
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black,
host_imgui_data(ctx));
}
}
if (stereoOutput) {
@@ -2231,6 +2255,14 @@ void record_frame_telemetry() {
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
Log.info("{}", gpuTiming);
}
if (const auto frameLog = g_frameLogCallback.load(std::memory_order_acquire)) {
char extra[512];
extra[0] = '\0';
frameLog(extra, sizeof(extra), elapsed.count(), windowFrames);
if (extra[0] != '\0') {
Log.info("{}", extra);
}
}
windowStart = now;
windowFrames = 0;
}
@@ -2242,7 +2274,7 @@ void record_frame_telemetry() {
// One complete frame-worker cycle. Desktop and headset interpolation both
// release the producer after sealing, before encoding their extra scene views.
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
const StereoSceneAnchor& sceneAnchor) noexcept {
ZoneScopedN("Frame worker cycle");
webgpu::fail_if_device_lost();
@@ -2254,7 +2286,7 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
auto stretchStarted = std::chrono::steady_clock::now();
{
std::lock_guard gpuLock(g_rendererGpuMutex);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
}
g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
stretchStarted = std::chrono::steady_clock::now();
@@ -2311,11 +2343,11 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
// Synchronous frame submission: seal, encode and present inline on the calling thread. Used when
// the frame worker is disabled (RenderDoc captures) and on the boot path.
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
const StereoSceneAnchor& sceneAnchor) noexcept {
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) noexcept {
ZoneScoped;
#ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost();
if (pumpEvents) {
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
window::pump_events();
}
gfx::SealedFrame sealedFrame;
@@ -2326,7 +2358,7 @@ void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
if (drainFifo) {
gx::fifo::drain();
}
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
}
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
@@ -2352,7 +2384,9 @@ bool begin_frame() noexcept {
ensure_frame_worker_started();
// SDL needs event pumping on the window-owning producer thread, and the worker passes
// pumpEvents=false, so keep it here even when the fast path returns early.
window::pump_events();
if (!g_hostEventPump.load(std::memory_order_acquire)) {
window::pump_events();
}
bool waitForSurfacePreparation = false;
#ifdef AURORA_ENABLE_GX
// A surface mutation can legitimately fail preparation, and optimistic success would let GX/ImGui
@@ -2402,7 +2436,7 @@ bool begin_frame() noexcept {
return prepared;
}
void end_frame(uint64_t contentTag) noexcept {
void end_frame(uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame) noexcept {
#ifdef AURORA_ENABLE_GX
webgpu::fail_if_device_lost();
#endif
@@ -2413,7 +2447,7 @@ void end_frame(uint64_t contentTag) noexcept {
g_pendingStereoLocalPlayerCount = 1;
g_pendingSceneAnchor = {};
if (!frame_worker_requested()) {
end_frame_impl(true, true, contentTag, sceneAnchor);
end_frame_impl(true, true, contentTag, sceneAnchor, std::move(hostImGuiFrame));
return;
}
@@ -2435,6 +2469,7 @@ void end_frame(uint64_t contentTag) noexcept {
g_frameWorker.ready.store(false, std::memory_order_release);
g_frameWorker.contentTag = contentTag;
g_frameWorker.sceneAnchor = sceneAnchor;
g_frameWorker.hostImGuiFrame = std::move(hostImGuiFrame);
g_frameWorker.jobPending = true;
g_frameWorker.prepareAllowed = false;
}
@@ -2537,8 +2572,34 @@ AuroraInfo aurora_initialize(int argc, char* argv[], const AuroraConfig* config)
void aurora_shutdown() { aurora::shutdown(); }
const AuroraEvent* aurora_update() { return aurora::update(); }
bool aurora_begin_frame() { return aurora::begin_frame(); }
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN); }
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag); }
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN, {}); }
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag, {}); }
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame) {
aurora::imgui::HostFramePtr frame;
if (imguiFrame != nullptr) {
auto* holder = static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
frame = std::move(*holder);
delete holder;
}
aurora::end_frame(contentTag, std::move(frame));
}
void aurora_set_host_event_pump(bool hostPumps) {
aurora::g_hostEventPump.store(hostPumps, std::memory_order_release);
}
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback) {
aurora::g_frameLogCallback.store(callback, std::memory_order_release);
}
extern "C" void aurora_imgui_host_frame_begin(void) {
#ifdef AURORA_ENABLE_GX
// ImGui's WebGPU backend creates its device objects lazily from new_frame.
std::lock_guard gpuLock(aurora::g_rendererGpuMutex);
#endif
aurora::imgui::host_frame_begin(aurora::window::get_window_size());
}
extern "C" void* aurora_imgui_host_frame_end(void) { return new aurora::imgui::HostFramePtr(aurora::imgui::host_frame_end()); }
extern "C" void aurora_imgui_host_frame_release(void* imguiFrame) {
delete static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
}
void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]) {
aurora::set_stereo_scene_anchor(anchorFromScene);
}
+76 -2
View File
@@ -28,7 +28,11 @@ static std::string g_imguiLog{};
static bool g_useSdlRenderer = false;
// Set once ImGui::Render() has produced this frame's draw data. Interpolation encodes up to four
// ImGui passes per frame, and every one of them used to rebuild the draw lists from scratch.
static bool g_frameDataBuilt = false;
static bool g_frameDataBuilt = true;
// Host-owned frames (see imgui.hpp). Once the host begins one, aurora never calls new_frame() or
// ImGui::Render() itself; the sealed frame carries the host's copy of the draw data instead.
static bool g_hostFrames = false;
static bool g_hostFrameOpen = false;
static std::vector<SDL_Texture*> g_sdlTextures;
static std::vector<wgpu::Texture> g_wgpuTextures;
@@ -179,7 +183,7 @@ void new_frame(const AuroraWindowSize& size) noexcept {
void render_frame_data() noexcept {
ZoneScoped;
if (g_frameDataBuilt) {
if (g_frameDataBuilt || g_hostFrames) {
return;
}
ImGui::Render();
@@ -190,6 +194,11 @@ void render_frame_data() noexcept {
void render(const wgpu::RenderPassEncoder& pass) noexcept {
ZoneScoped;
if (g_hostFrames) {
// The shared context's draw data belongs to the host's current frame now;
// a sealed frame without a host copy has nothing safe to draw.
return;
}
render_frame_data();
auto* data = ImGui::GetDrawData();
@@ -205,6 +214,71 @@ void render(const wgpu::RenderPassEncoder& pass) noexcept {
}
}
struct HostFrame {
ImDrawData data{};
std::vector<ImDrawList*> lists;
~HostFrame() {
for (ImDrawList* list : lists) {
IM_DELETE(list);
}
}
};
void host_frame_begin(const AuroraWindowSize& size) noexcept {
g_hostFrames = true;
if (g_hostFrameOpen) {
return;
}
if (!g_frameDataBuilt) {
// aurora started this frame itself before the host took over: adopt it.
g_hostFrameOpen = true;
return;
}
new_frame(size);
g_hostFrameOpen = true;
}
HostFramePtr host_frame_end() noexcept {
ZoneScoped;
if (!g_hostFrameOpen) {
host_frame_begin(window::get_window_size());
}
ImGui::Render();
ImDrawData* source = ImGui::GetDrawData();
source->FramebufferScale = ImGui::GetIO().DisplayFramebufferScale;
auto frame = std::make_shared<HostFrame>();
frame->data = *source;
frame->data.CmdLists.clear();
frame->lists.reserve(static_cast<size_t>(source->CmdListsCount));
for (int i = 0; i < source->CmdListsCount; ++i) {
const ImDrawList* src = source->CmdLists[i];
ImDrawList* copy = IM_NEW(ImDrawList)(src->_Data);
copy->CmdBuffer = src->CmdBuffer;
copy->IdxBuffer = src->IdxBuffer;
copy->VtxBuffer = src->VtxBuffer;
copy->Flags = src->Flags;
frame->lists.push_back(copy);
frame->data.CmdLists.push_back(copy);
}
g_hostFrameOpen = false;
g_frameDataBuilt = true;
return frame;
}
bool host_frames_active() noexcept { return g_hostFrames; }
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept { return &frame.data; }
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept {
ZoneScoped;
if (g_useSdlRenderer || data == nullptr) {
return;
}
pass.PushDebugGroup("Aurora: Dear Imgui");
ImGui_ImplWGPU_RenderDrawData(const_cast<ImDrawData*>(data), pass.Get());
pass.PopDebugGroup();
}
StereoOverlay latch_stereo_overlay() noexcept {
std::lock_guard lock(g_stereoOverlayMutex);
return g_stereoOverlay;
+14
View File
@@ -1,6 +1,7 @@
#pragma once
#include <aurora/event.h>
#include <memory>
union SDL_Event;
struct ImDrawData;
@@ -32,4 +33,17 @@ StereoOverlay latch_stereo_overlay() noexcept;
// uniform for every pass, so a pass whose display size differs from the desktop's must be submitted
// before the next pass is recorded.
bool render_draw_data(const wgpu::RenderPassEncoder& pass, ImDrawData* data) noexcept;
// Host-owned ImGui frames. The host starts each frame on its own thread with host_frame_begin() and
// closes it with host_frame_end(), which renders the frame and copies its draw data out of the shared
// context. The copy is what the sealed frame replays, so the host may start the next frame while the
// worker still encodes this one, and aurora stops starting frames itself once the host has begun one.
struct HostFrame;
using HostFramePtr = std::shared_ptr<HostFrame>;
void host_frame_begin(const AuroraWindowSize& size) noexcept;
HostFramePtr host_frame_end() noexcept;
bool host_frames_active() noexcept;
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept;
// Renders a host frame's copied draw data in place of the shared context's.
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept;
} // namespace aurora::imgui
+37 -1
View File
@@ -588,7 +588,13 @@ the app:
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` presses the settings panel's button (left Y, or both thumbsticks as a gamepad), opening or closing it (see `OPENXR.md`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A third line reports the GX thread's command ring (records, waits, busy share). A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
A `Config.toml` written with `adb push` (or `sed -i` in `adb shell`) belongs
to the shell user afterwards, and the app then fails every save with EACCES
(the launcher logs `GameStorage.prepare ... open failed`). `chmod 664` on the
pushed file gives the app's group write access back; a file the app created
itself never has the problem.
The injector makes headset tests possible with nobody wearing the headset.
Keep the display awake, drive the menus, then take a compositor screenshot:
@@ -687,6 +693,36 @@ plus mod code, 9% GX HLE, 6% FIFO decode, 4% memory copies, 3.5% dispatch and
the rest. What remains on such tracks is the game's own code plus the mod's,
which no host change shrinks; a GX thread could move about 20% of it.
That GX thread exists now (`runtime/include/gx_thread.h`, `[video] gx_thread`,
on by default on Android and opt-in elsewhere). Every GX HLE override is split
into a game-thread front, which keeps the guest-visible side effects (GXData
shadow registers, the getters, display-list recording, the texture meta table),
and a `_gx` back holding the aurora work and the parser state, posted through
one ordered 16 MiB command ring; immediate-mode gather-pipe bytes travel as
8 KiB chunks in call order. The hazard rule follows the hardware: whatever the
SDK copied into the FIFO at call time (matrices, projection, colours, light
objects, copy filters, layout quads, texture object registers) is snapshotted
when posted, and whatever the GP read from memory when it reached the command
(display lists, vertex arrays, indexed matrices, texture data) is read when the
GX thread executes it, so `GXDrawDone` drains the ring and the frame's
schedule, first-person anchor and policy tag are latched into the present
record on the game thread. The desktop overlay became a game-thread-owned
ImGui frame whose draw data Aurora copies per sealed frame, which also removed
the frame-worker join `GXCopyDisp` used to make. With `fpslog` on, a third
line reports the ring: records and bytes per frame, the game thread's waits
for ring space and in drains, the GX thread's busy share and any exceptions
it caught. A texture or matrix that is wrong only with the thread on is a
hazard-rule violation (a front reading guest memory the game rewrites before
the GX thread runs, or a back writing guest memory). Measured on the same
automated Grand Prix start at `render_scale` 0.75, same build, switched by the
config key: with the thread off the crowded first half minute ran at 52 to
56 fps before settling at 60; with it on the same stretch ran at 56.5 in the
window that includes the countdown and 60.0 in every window after, while the
ring carried 4.5k to 6.2k records (250 to 380 KiB) per frame, the game thread
waited under 0.1 ms per frame in its two `GXDrawDone` drains and never for
ring space, and the GX thread was 25 to 40% busy. Retro Rewind's menus were
unaffected (prewarm 5.2 s, 60 fps).
Verified on device since: the menus on the virtual screen, controller input
(the user has driven races), and an immersive Grand Prix start with all 12
racers rendering correctly. Not yet verified: stereo comfort and scale,
+6 -1
View File
@@ -3,6 +3,7 @@
#include "settings_overlay.h"
#include "runtime_config.h"
#include "fiber_manager.h"
#include "gx_thread.h"
#include <aurora/aurora.h>
#include <aurora/event.h>
@@ -142,7 +143,11 @@ inline bool BeginAuroraFrame() {
if (!aurora_begin_frame()) {
return false;
}
ApplyPendingMkwDynamicAspectSurface();
// The viewport policy writes guest memory (EGG screen records), so the GX
// thread leaves it to the game thread's present path.
if (!GxThread::IsGxThread()) {
ApplyPendingMkwDynamicAspectSurface();
}
return true;
}
+106
View File
@@ -0,0 +1,106 @@
#pragma once
// GX thread: the host side of the GX pipeline (state tracking, FIFO parsing,
// display-list scanning, texture object resolution and every aurora GX call)
// runs on its own thread, fed by an ordered command ring the game thread posts
// to. Each GX HLE override is split into a game-thread front (guest-visible
// side effects: shadow registers, getters, display-list recording) and a
// GX-thread back (the aurora work). When the thread is disabled, Post() runs
// the back inline, so the split is behaviour-preserving in both modes.
//
// Data hazards follow the hardware: anything the GX library copies into the
// FIFO at call time (immediate-mode vertices, matrices, colours, light objects,
// texture object registers) is snapshotted at post time; anything the GP reads
// from memory when it reaches the command (display lists, vertex arrays,
// indexed matrices, texture data) is read when the GX thread executes it.
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <string>
namespace GxThread {
// Decides whether posted work runs on the GX thread. Read once, before GXInit.
void Configure(bool enabled);
bool Enabled() noexcept;
// True on the consumer thread.
bool IsGxThread() noexcept;
void Start();
// Executes everything posted so far and joins the thread.
void Stop();
// Game thread: blocks until every posted record has executed (GXDrawDone).
void Drain();
// Publishes the open immediate-mode FIFO chunk. Post() does this itself.
void FlushFifo();
// Invoked at bounded intervals while the game thread blocks in Drain() or on
// a full ring, so guest timing (VI retraces, OS alarms) keeps running.
void SetWaitCallback(void (*callback)());
// Native thread id of the consumer (Android: gettid), 0 until started.
uint32_t NativeThreadId() noexcept;
// One line for the frame-rate log; resets the window counters.
std::string FormatStatsAndReset(double windowSeconds, uint32_t frames);
// Immediate-mode write-gather bytes. Only valid when Enabled().
void PostFifoWord(uint32_t value, uint32_t sizeBytes);
void PostFifoBytes(const uint8_t* data, uint32_t sizeBytes);
namespace detail {
using Invoke = void (*)(const uint8_t* payload, uint32_t payloadBytes);
void PostRecord(Invoke invoke, const void* payload, uint32_t payloadBytes);
template <typename... Ts>
struct Pack;
template <>
struct Pack<> {
template <typename F, typename... Prev>
void Call(F f, Prev... prev) const {
f(prev...);
}
};
template <typename T, typename... Ts>
struct Pack<T, Ts...> {
T head;
Pack<Ts...> tail;
template <typename F, typename... Prev>
void Call(F f, Prev... prev) const {
tail.Call(f, prev..., head);
}
};
template <typename... Ts>
struct BuildPack;
template <>
struct BuildPack<> {
static Pack<> Make() { return {}; }
};
template <typename T, typename... Ts>
struct BuildPack<T, Ts...> {
static Pack<T, Ts...> Make(T head, Ts... tail) {
return Pack<T, Ts...>{head, BuildPack<Ts...>::Make(tail...)};
}
};
template <typename R, typename... Params>
struct CallRecord {
R (*fn)(Params...);
Pack<Params...> args;
};
template <typename R, typename... Params>
void InvokeCall(const uint8_t* payload, uint32_t) {
CallRecord<R, Params...> record;
std::memcpy(&record, payload, sizeof(record));
record.args.Call(record.fn);
}
} // namespace detail
// Posts fn(args...) to the GX thread, or runs it now when the thread is off.
// Parameters must be trivially copyable values (no pointers into guest memory
// that the game may rewrite before the GX thread reads them).
template <typename R, typename... Params, typename... Args>
inline void Post(R (*fn)(Params...), Args&&... args) {
if (!Enabled()) {
fn(static_cast<Params>(args)...);
return;
}
detail::CallRecord<R, Params...> record{fn, detail::BuildPack<Params...>::Make(static_cast<Params>(args)...)};
detail::PostRecord(&detail::InvokeCall<R, Params...>, &record, sizeof(record));
}
} // namespace GxThread
+16
View File
@@ -47,6 +47,7 @@ struct RuntimeUserConfig {
std::optional<bool> textureReplacements;
std::optional<bool> textureDumps;
std::optional<bool> showFps;
std::optional<bool> gxThread;
std::optional<uint32_t> disabledPostProcessingPaths;
std::optional<bool> vrEnabled;
std::optional<bool> vrRequired;
@@ -410,6 +411,10 @@ inline void EnsureConfigFile() {
"skip_unready_pipelines = true\n"
"disable_copy_filter = true\n"
"show_fps = true\n"
"# Run the host side of the GX pipeline (state tracking, FIFO parsing,\n"
"# texture uploads) on its own thread. On by default on the Quest, where\n"
"# the game thread is the bottleneck; opt-in elsewhere.\n"
"# gx_thread = true\n"
"# Dolphin-style custom textures. When enabled, the renderer indexes\n"
"# texture_replacements/ next to this file at startup and substitutes\n"
"# any tex1_<W>x<H>_<hash>[_<tlut hash>]_<format>.dds or .png it finds\n"
@@ -641,6 +646,7 @@ inline RuntimeUserConfig ParseConfigDocument(const toml::value& document) {
config.skipUnreadyPipelines = FindConfigValue<bool>(document, "video", "skip_unready_pipelines");
config.disableCopyFilter = FindConfigValue<bool>(document, "video", "disable_copy_filter");
config.showFps = FindConfigValue<bool>(document, "video", "show_fps");
config.gxThread = FindConfigValue<bool>(document, "video", "gx_thread");
config.textureReplacements = FindConfigValue<bool>(document, "video", "texture_replacements");
config.textureDumps = FindConfigValue<bool>(document, "video", "texture_dumps");
if (auto value = FindConfigUint(document, "video", "disabled_post_processing_paths");
@@ -1294,6 +1300,16 @@ inline bool ShowFps(bool fallback = true) {
return Get().showFps.value_or(fallback);
}
// Runs the host side of the GX pipeline on its own thread (gx_thread.h). On by
// default on the Quest, where the game thread is the bottleneck; opt-in elsewhere.
inline bool GxThread() {
#if defined(__ANDROID__)
return Get().gxThread.value_or(true);
#else
return Get().gxThread.value_or(false);
#endif
}
inline bool TextureReplacements(bool fallback = false) {
return Get().textureReplacements.value_or(fallback);
}
+46 -26
View File
@@ -62,69 +62,84 @@ void InvalidateEfbCopyDestinationsForRange(uint32_t addr, uint32_t size) {
// Preserve FIFO ordering: the destroy command is emitted before any later
// texture load that can consume the freshly flushed RAM bytes.
for (const uint32_t copyAddr : retired) {
GXDestroyCopyTex(GuestToHostPtr(copyAddr));
GxThread::Post(&GxHostDestroyCopyTex_gx, copyAddr);
}
}
void GxHostDestroyCopyTex_gx(uint32_t copyAddr) { GXDestroyCopyTex(GuestToHostPtr(copyAddr)); }
// ============================================================================
// Display Copy Source/Destination
// ============================================================================
extern "C" void GX__SetDispCopySrc_8016f438(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
static void GX__SetDispCopySrc_8016f438_gx(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
GXSetDispCopySrc((u16)l, (u16)t, (u16)w, (u16)h);
}
PPC_NATIVE_OVERRIDE_VOID(8016f438, GX__SetDispCopySrc_8016f438, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
GX_DEFERRED_OVERRIDE_VOID(8016f438, GX__SetDispCopySrc_8016f438, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
extern "C" void GX__SetDispCopyDst_8016f4b8(uint32_t w, uint32_t h) { GXSetDispCopyDst((u16)w, (u16)h); }
PPC_NATIVE_OVERRIDE_VOID(8016f4b8, GX__SetDispCopyDst_8016f4b8, (uint32_t w, uint32_t h), (w, h));
static void GX__SetDispCopyDst_8016f4b8_gx(uint32_t w, uint32_t h) { GXSetDispCopyDst((u16)w, (u16)h); }
GX_DEFERRED_OVERRIDE_VOID(8016f4b8, GX__SetDispCopyDst_8016f4b8, (uint32_t w, uint32_t h), (w, h));
// ============================================================================
// Texture Copy Source/Destination
// ============================================================================
extern "C" void GX__SetTexCopySrc_8016f478(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
static void GX__SetTexCopySrc_gx(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
GXSetTexCopySrc((u16)l, (u16)t, (u16)w, (u16)h);
}
extern "C" void GX__SetTexCopySrc_8016f478(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
GxThread::Post(&GX__SetTexCopySrc_gx, l, t, w, h);
g_texCopyState.srcLeft=(u16)l; g_texCopyState.srcTop=(u16)t;
g_texCopyState.srcWidth=(u16)w; g_texCopyState.srcHeight=(u16)h;
}
PPC_NATIVE_OVERRIDE_VOID(8016f478, GX__SetTexCopySrc_8016f478, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
extern "C" void GX__SetTexCopyDst_8016f4dc(uint32_t w, uint32_t h, uint32_t f, uint32_t m) {
static void GX__SetTexCopyDst_gx(uint32_t w, uint32_t h, uint32_t f, uint32_t m) {
GXSetTexCopyDst((u16)w, (u16)h, (GXTexFmt)f, (GXBool)m);
}
extern "C" void GX__SetTexCopyDst_8016f4dc(uint32_t w, uint32_t h, uint32_t f, uint32_t m) {
GxThread::Post(&GX__SetTexCopyDst_gx, w, h, f, m);
g_texCopyState.dstWidth=(u16)w; g_texCopyState.dstHeight=(u16)h;
g_texCopyState.dstFormat=f; g_texCopyState.dstMipmap=m;
}
PPC_NATIVE_OVERRIDE_VOID(8016f4dc, GX__SetTexCopyDst_8016f4dc, (uint32_t w, uint32_t h, uint32_t f, uint32_t m), (w, h, f, m));
struct GxCopyFilterSnapshot {
uint8_t sp[12][2];
uint8_t vfb[7];
};
static void GX__SetCopyFilter_gx(uint32_t aa, uint32_t vf, GxCopyFilterSnapshot filter) {
GXSetCopyFilter((GXBool)aa, filter.sp, (GXBool)vf, filter.vfb);
}
extern "C" void GX__SetCopyFilter_8016fa40(uint32_t aa, uint32_t spa, uint32_t vf, uint32_t vfa) {
uint8_t sp[12][2]={}, vfb[7]={};
if(spa) std::memcpy(sp, GuestToHostPtr(spa, 24), 24);
if(vfa) std::memcpy(vfb, GuestToHostPtr(vfa, 7), 7);
GXSetCopyFilter((GXBool)aa, sp, (GXBool)vf, vfb);
GxCopyFilterSnapshot filter{};
if(spa) std::memcpy(filter.sp, GuestToHostPtr(spa, 24), 24);
if(vfa) std::memcpy(filter.vfb, GuestToHostPtr(vfa, 7), 7);
GxThread::Post(&GX__SetCopyFilter_gx, aa, vf, filter);
}
PPC_NATIVE_OVERRIDE_VOID(8016fa40, GX__SetCopyFilter_8016fa40, (uint32_t aa, uint32_t spa, uint32_t vf, uint32_t vfa), (aa, spa, vf, vfa));
extern "C" void GX__SetDispCopyGamma_8016fc24(uint32_t g) { GXSetDispCopyGamma((GXGamma)g); }
PPC_NATIVE_OVERRIDE_VOID(8016fc24, GX__SetDispCopyGamma_8016fc24, (uint32_t g), (g));
static void GX__SetDispCopyGamma_8016fc24_gx(uint32_t g) { GXSetDispCopyGamma((GXGamma)g); }
GX_DEFERRED_OVERRIDE_VOID(8016fc24, GX__SetDispCopyGamma_8016fc24, (uint32_t g), (g));
// ============================================================================
// Copy Execution
// ============================================================================
extern "C" void GX__CopyDisp_8016fc38(uint32_t da, uint32_t c) {
static void GX__CopyDisp_gx(uint32_t da, uint32_t c) {
EnsureAuroraFrameActive();
// GX copies are FIFO-ordered on hardware. Drain submitted draws before
// resolving the EFB so high-level copies see the same contents.
GXDrawDone();
GXCopyDisp(GuestToHostPtr(da), (GXBool)c);
// No second GXDrawDone here: the frame-worker wait below is for the DONE
// phase, which strictly subsumes the drain this call would perform.
}
extern "C" void GX__CopyDisp_8016fc38(uint32_t da, uint32_t c) {
GxThread::Post(&GX__CopyDisp_gx, da, c);
++g_gxFrameCount;
VI_HLE_SetXfbReady(da);
// Present immediately so post-copy draws don't leak into this frame. Join at the DONE phase
// (not the cheaper SEALED phase GXDrawDone waits for) because ImGui's draw lists, owned by
// Aurora's render worker, replay during encode; aurora_end_frame would join here anyway.
aurora_wait_for_frame_worker();
// Present immediately so post-copy draws don't leak into this frame. The
// overlay draws into the game thread's own ImGui frame, whose draw data
// the seal copies, so no frame-worker join is needed here.
settings_overlay::Draw();
// Seal, pace to the VI retrace boundary (Aurora renders the sealed frame
// during the wait), and pre-warm the next frame.
@@ -134,14 +149,15 @@ extern "C" void GX__CopyDisp_8016fc38(uint32_t da, uint32_t c) {
PPC_NATIVE_OVERRIDE_VOID(8016fc38, GX__CopyDisp_8016fc38, (uint32_t da, uint32_t c), (da, c));
extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) {
static void GX__CopyTex_gx(uint32_t da, uint32_t c, uint32_t srcLeft, uint32_t srcTop, uint32_t srcWidth,
uint32_t srcHeight) {
EnsureAuroraFrameActive();
// Match GX FIFO ordering: texture copies observe all prior draws.
GXDrawDone();
const uint16_t rawSrcLeft = g_texCopyState.srcLeft;
const uint16_t rawSrcTop = g_texCopyState.srcTop;
const uint16_t rawSrcWidth = g_texCopyState.srcWidth;
const uint16_t rawSrcHeight = g_texCopyState.srcHeight;
const uint16_t rawSrcLeft = (uint16_t)srcLeft;
const uint16_t rawSrcTop = (uint16_t)srcTop;
const uint16_t rawSrcWidth = (uint16_t)srcWidth;
const uint16_t rawSrcHeight = (uint16_t)srcHeight;
// Keep the source in guest EFB coordinates. Aurora maps it to the scaled
// EFB exactly once, matching Dolphin's ConvertEFBRectangle path.
@@ -153,9 +169,13 @@ extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) {
// auto-downloaded, so guest reads see stale RAM; call aurora_flush_efb_copies_to_ram if a
// copy needs reading back.
GXCopyTex(GuestToHostPtr(da), (GXBool)c);
GXSetTexCopySrc(rawSrcLeft, rawSrcTop, rawSrcWidth, rawSrcHeight);
}
extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) {
GxThread::Post(&GX__CopyTex_gx, da, c, g_texCopyState.srcLeft, g_texCopyState.srcTop,
g_texCopyState.srcWidth, g_texCopyState.srcHeight);
RememberEfbCopyDestination(
da, GXGetTexBufferSize(g_texCopyState.dstWidth, g_texCopyState.dstHeight,
g_texCopyState.dstFormat, GX_FALSE, 0));
GXSetTexCopySrc(rawSrcLeft, rawSrcTop, rawSrcWidth, rawSrcHeight);
}
PPC_NATIVE_OVERRIDE_VOID(8016fd74, GX__CopyTex_8016fd74, (uint32_t da, uint32_t c), (da, c));
+37 -38
View File
@@ -128,7 +128,7 @@ static inline void AppendLytQuadVertices(uint8_t* packet, uint32_t& pos, float x
AppendLytQuadVertex(packet, pos, x0, y1, texCoordAddr, texCoordCount, 16, colors, 2);
}
static bool CanSubmitLytDrawDirect(int texCoordCount, const uint32_t* colors) {
static bool CanSubmitLytDrawDirect(int texCoordCount, bool hasColors) {
if (texCoordCount < 0 || texCoordCount > 8) {
return false;
}
@@ -139,7 +139,6 @@ static bool CanSubmitLytDrawDirect(int texCoordCount, const uint32_t* colors) {
return false;
}
const bool hasColors = colors != nullptr;
const auto& clrFmt = g_hleGxState.vtxAttrFmt[GX_VTXFMT0][GX_VA_CLR0];
if (hasColors) {
if (g_hleGxState.vtxDesc[GX_VA_CLR0] != GX_DIRECT ||
@@ -179,33 +178,30 @@ static bool CanSubmitLytDrawDirect(int texCoordCount, const uint32_t* colors) {
return true;
}
static bool SubmitLytDrawDirect(float x0, float y0, float x1, float y1, int texCoordCount,
uint32_t texCoordAddr, const uint32_t* colors) {
if (!CanSubmitLytDrawDirect(texCoordCount, colors)) {
return false;
// One nw4r::lyt quad as a GX draw packet (3-byte header, 4 vertices), built on
// the game thread from the guest's layout data and submitted on the GX thread,
// whose descriptor state decides between the raw-draw fast path and the packet
// parser.
struct GxLytQuadPacket {
uint32_t bytes = 0;
int32_t texCoordCount = 0;
bool hasColors = false;
uint8_t data[3u + 4u * (8u + 4u + 8u * 8u)];
};
static void GxLytQuad_gx(GxLytQuadPacket packet) {
if (CanSubmitLytDrawDirect(packet.texCoordCount, packet.hasColors)) {
EnsureAuroraFrameActive();
ApplyAuroraVtxDesc();
ApplyAuroraVtxAttrFmtForDisplayList(GX_VTXFMT0, false);
EnsureDefaultGxAlphaCompare();
if (aurora::gx::fifo::submit_raw_draw(GX_QUADS, GX_VTXFMT0, packet.data + 3, 4, packet.bytes - 3u)) {
GXMarkFrameWork();
SyncAppliedVtxStateFromHleReal();
return;
}
}
EnsureAuroraFrameActive();
ApplyAuroraVtxDesc();
ApplyAuroraVtxAttrFmtForDisplayList(GX_VTXFMT0, false);
EnsureDefaultGxAlphaCompare();
std::array<uint8_t, 4u * (8u + 4u + 8u * 8u)> vertices{};
uint32_t pos = 0;
AppendLytQuadVertices(vertices.data(), pos, x0, y0, x1, y1, texCoordAddr, texCoordCount, colors);
if (!aurora::gx::fifo::submit_raw_draw(GX_QUADS, GX_VTXFMT0, vertices.data(), 4, pos)) {
return false;
}
GXMarkFrameWork();
SyncAppliedVtxStateFromHleReal();
return true;
SubmitLytDrawPacket(packet.data, packet.bytes);
}
static inline void EmitLytDrawQuad(uint32_t posAddr, uint32_t sizeAddr, int texCoordCount,
@@ -217,10 +213,6 @@ static inline void EmitLytDrawQuad(uint32_t posAddr, uint32_t sizeAddr, int texC
const float x1 = static_cast<float>(x0 + Memory::ReadFloat32(sizeAddr));
const float y1 = static_cast<float>(y0 - Memory::ReadFloat32(sizeAddr + 4));
if (SubmitLytDrawDirect(x0, y0, x1, y1, texCoordCount, texCoordAddr, colors)) {
return;
}
// GX has exactly 8 texture coordinates, so nw4r::lyt cannot ask for more.
// The fixed packet buffer below is sized for that maximum; bail rather than
// overrun it if the guest ever hands us something else.
@@ -228,12 +220,15 @@ static inline void EmitLytDrawQuad(uint32_t posAddr, uint32_t sizeAddr, int texC
return;
}
std::array<uint8_t, 3u + 4u * (8u + 4u + 8u * 8u)> packet{};
GxLytQuadPacket packet{};
packet.texCoordCount = texCoordCount;
packet.hasColors = colors != nullptr;
uint32_t pos = 0;
packet[pos++] = GX_DRAW_QUADS_CMD | GX_VTXFMT0;
BigEndian::Append16(packet.data(), pos, 4);
AppendLytQuadVertices(packet.data(), pos, x0, y0, x1, y1, texCoordAddr, texCoordCount, colors);
SubmitLytDrawPacket(packet.data(), pos);
packet.data[pos++] = GX_DRAW_QUADS_CMD | GX_VTXFMT0;
BigEndian::Append16(packet.data, pos, 4);
AppendLytQuadVertices(packet.data, pos, x0, y0, x1, y1, texCoordAddr, texCoordCount, colors);
packet.bytes = pos;
GxThread::Post(&GxLytQuad_gx, packet);
}
static uint32_t SubmitDLVertex(const uint8_t* ptr, GXVtxFmt vtxfmt, const GXAttrType* sourceVtxDesc) {
@@ -1293,7 +1288,8 @@ extern "C" void GxNotifyDisplayListMemoryWrite(uint32_t addr, uint32_t size) {
GxGuestWrite::NotifyWrite(addr, size);
}
extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes) {
// The list itself is read when the GX thread reaches the call, as the GP does.
void GX__CallDisplayList_gx(uint32_t listAddr, uint32_t nbytes) {
if (nbytes == 0 || listAddr == 0) return;
try {
const uint8_t* list = static_cast<const uint8_t*>(GuestToHostPtr(listAddr, nbytes));
@@ -1487,6 +1483,9 @@ extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes)
} catch (...) {}
}
extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes) {
GxThread::Post(&GX__CallDisplayList_gx, listAddr, nbytes);
}
PPC_NATIVE_OVERRIDE_VOID(80172F64, GX__CallDisplayList_80172f64, (uint32_t listAddr, uint32_t nbytes), (listAddr, nbytes));
extern "C" void nw4r__lyt__detail__DrawQuad_800847c0(CpuContext* ctx) {
+3 -3
View File
@@ -79,10 +79,10 @@ extern "C" void EGG__LightTexture__SetupTevFinish_HLE_8022e2bc(CpuContext* ctx)
if (stageCount != 0) {
uint32_t remainder = tevCount % stageCount;
if (remainder > 0) {
const GXColor black{0, 0, 0, 255};
constexpr uint32_t kBlack = 0x000000FFu;
while (remainder < stageCount) {
GXSetTevColor(static_cast<GXTevRegID>(remainder + 1), black);
GXSetTevKColor(static_cast<GXTevKColorID>(remainder), black);
GxThread::Post(&GX__SetTevColor_gx, remainder + 1, kBlack);
GxThread::Post(&GX__SetTevKColor_gx, remainder, kBlack);
++remainder;
}
+49 -12
View File
@@ -2,6 +2,7 @@
#include "gx_stream_common.h"
#include "gx_cp_decode.h"
#include "isa/big_endian.h"
#include "runtime_log.h"
// Opcode constants and the stream helpers this file shares with gx_dl.cpp /
// gx_vertex.cpp; see gx_stream_common.h.
@@ -346,7 +347,9 @@ void SubmitAttribute(GXAttr attr, float* comps, const VtxAttrFmt& fmt, const u32
// `val` is a raw big-endian bit pattern: the FIFO stream is type-agnostic, and
// the float entry point converts before it gets here.
void HleFifoWrite(u32 val, uint32_t sizeBytes) {
const bool recordOnly = IsDisplayListActive();
// Display-list recording is game-thread state; bytes only reach the GX
// thread when nothing is being recorded, so it must not consult it.
const bool recordOnly = !GxThread::IsGxThread() && IsDisplayListActive();
if (recordOnly) {
WriteDisplayListData(val, sizeBytes);
return;
@@ -480,7 +483,7 @@ void HleFifoWrite(u32 val, uint32_t sizeBytes) {
const uint32_t listSize = ReadBE32(data + 5);
if (!consumeBytes(9, sink)) break;
if (listAddr != 0 && listSize > 0) {
GX__CallDisplayList_80172f64(listAddr, listSize);
GX__CallDisplayList_gx(listAddr, listSize);
}
continue;
}
@@ -682,7 +685,8 @@ static uint32_t ApplyFifoPacketsDirect(const uint8_t* data, uint32_t sizeBytes)
while (offset < sizeBytes) {
// Re-tested per packet, not once per burst: nothing currently re-enters GX HLE mid-walk,
// but if it ever does, breaking here just hands the remainder to the ring.
if (IsDisplayListActive() || g_hleGxState.inBegin || g_hleGxState.fifoByteCount != 0) {
if ((!GxThread::IsGxThread() && IsDisplayListActive()) || g_hleGxState.inBegin ||
g_hleGxState.fifoByteCount != 0) {
break;
}
@@ -775,17 +779,50 @@ static bool WriteDisplayListBurst(const uint8_t* data, uint32_t sizeBytes) {
return true;
}
extern "C" void GX_HLE_FIFO_WriteBurst(const uint8_t* data, uint32_t sizeBytes) {
if (data == nullptr || sizeBytes == 0) {
return;
}
if (IsDisplayListActive() && WriteDisplayListBurst(data, sizeBytes)) {
return;
}
// GX-thread side of the write-gather pipe (or inline when the thread is off).
extern "C" void GxFifoConsumeBytes(const uint8_t* data, uint32_t sizeBytes) {
const uint32_t applied = ApplyFifoPacketsDirect(data, sizeBytes);
if (applied < sizeBytes) {
HleFifoWriteBurstChunked(data + applied, sizeBytes - applied);
}
}
// Game-thread fronts of the write-gather pipe. Display-list recording is
// resolved here (it writes guest memory); everything else is parsed now or
// posted to the GX thread as raw bytes, in call order.
static inline void GxFifoFrontWrite(u32 val, uint32_t sizeBytes) {
if (IsDisplayListActive()) {
WriteDisplayListData(val, sizeBytes);
return;
}
if (GxThread::Enabled()) {
GxThread::PostFifoWord(val, sizeBytes);
return;
}
HleFifoWrite(val, sizeBytes);
}
extern "C" void GX_HLE_FIFO_WriteFloat(float val) {
u32 raw; std::memcpy(&raw, &val, 4);
try { GxFifoFrontWrite(raw, 4); } catch (...) { RT_LOGF(RT_TAG_GX, "FIFO write float failed\n"); }
}
extern "C" void GX_HLE_FIFO_Write32(uint32_t val) { GxFifoFrontWrite(val, 4); }
extern "C" void GX_HLE_FIFO_Write16(uint16_t val) { GxFifoFrontWrite(static_cast<u32>(val), 2); }
extern "C" void GX_HLE_FIFO_Write8(uint8_t val) { GxFifoFrontWrite(static_cast<u32>(val), 1); }
extern "C" void GX_HLE_FIFO_WriteBurst(const uint8_t* data, uint32_t sizeBytes) {
if (data == nullptr || sizeBytes == 0) {
return;
}
if (IsDisplayListActive()) {
if (!WriteDisplayListBurst(data, sizeBytes)) {
HleFifoWriteBurstChunked(data, sizeBytes);
}
return;
}
if (GxThread::Enabled()) {
GxThread::PostFifoBytes(data, sizeBytes);
return;
}
GxFifoConsumeBytes(data, sizeBytes);
}
+4 -1
View File
@@ -68,7 +68,10 @@ void BeginNextAuroraFrameWithRetry(std::chrono::milliseconds timeout) {
const auto deadline = std::chrono::steady_clock::now() + timeout;
uint32_t attempts = 0;
while (std::chrono::steady_clock::now() < deadline) {
UpdateAuroraAndProcessEvents();
// SDL is pumped by the window's thread; the GX thread only retries the begin.
if (!GxThread::IsGxThread()) {
UpdateAuroraAndProcessEvents();
}
++attempts;
if (BeginAuroraFrame()) {
g_auroraFrameActive.store(true, std::memory_order_release);
+21 -14
View File
@@ -5,22 +5,29 @@
// Indirect Texture Stages
// ============================================================================
extern "C" void GX__SetNumIndStages_80171b38(uint32_t n) { GXSetNumIndStages((u8)n); }
PPC_NATIVE_OVERRIDE_VOID(80171b38, GX__SetNumIndStages_80171b38, (uint32_t n), (n));
static void GX__SetNumIndStages_80171b38_gx(uint32_t n) { GXSetNumIndStages((u8)n); }
GX_DEFERRED_OVERRIDE_VOID(80171b38, GX__SetNumIndStages_80171b38, (uint32_t n), (n));
extern "C" void GX__SetIndTexOrder_80171a6c(uint32_t s, uint32_t c, uint32_t m) {
static void GX__SetIndTexOrder_80171a6c_gx(uint32_t s, uint32_t c, uint32_t m) {
GXSetIndTexOrder((GXIndTexStageID)s, (GXTexCoordID)(c==0xFFu?0:c), (GXTexMapID)(m==0xFFu?0:m));
}
PPC_NATIVE_OVERRIDE_VOID(80171a6c, GX__SetIndTexOrder_80171a6c, (uint32_t s, uint32_t c, uint32_t m), (s, c, m));
GX_DEFERRED_OVERRIDE_VOID(80171a6c, GX__SetIndTexOrder_80171a6c, (uint32_t s, uint32_t c, uint32_t m), (s, c, m));
extern "C" void GX__SetIndTexCoordScale_80171968(uint32_t s, uint32_t ss, uint32_t ts) {
static void GX__SetIndTexCoordScale_80171968_gx(uint32_t s, uint32_t ss, uint32_t ts) {
GXSetIndTexCoordScale((GXIndTexStageID)s, (GXIndTexScale)ss, (GXIndTexScale)ts);
}
PPC_NATIVE_OVERRIDE_VOID(80171968, GX__SetIndTexCoordScale_80171968, (uint32_t s, uint32_t ss, uint32_t ts), (s, ss, ts));
GX_DEFERRED_OVERRIDE_VOID(80171968, GX__SetIndTexCoordScale_80171968, (uint32_t s, uint32_t ss, uint32_t ts), (s, ss, ts));
struct GxIndTexMtxSnapshot {
float m[6];
};
static void GX__SetIndTexMtx_gx(uint32_t id, GxIndTexMtxSnapshot mtx, uint32_t se) {
GXSetIndTexMtx((GXIndTexMtxID)id, mtx.m, (s8)se);
}
extern "C" void GX__SetIndTexMtx_80171814(uint32_t id, uint32_t ma, uint32_t se) {
float m[6]; for(int i=0; i<6; ++i) m[i]=Memory::ReadFloat32(ma+i*4);
GXSetIndTexMtx((GXIndTexMtxID)id, m, (s8)se);
GxIndTexMtxSnapshot mtx{};
for(int i=0; i<6; ++i) mtx.m[i]=Memory::ReadFloat32(ma+i*4);
GxThread::Post(&GX__SetIndTexMtx_gx, id, mtx, se);
}
PPC_NATIVE_OVERRIDE_VOID(80171814, GX__SetIndTexMtx_80171814, (uint32_t id, uint32_t ma, uint32_t se), (id, ma, se));
@@ -28,18 +35,18 @@ PPC_NATIVE_OVERRIDE_VOID(80171814, GX__SetIndTexMtx_80171814, (uint32_t id, uint
// TEV Indirect Texture Control
// ============================================================================
extern "C" void GX__SetTevDirect_80171b58(uint32_t s) { GXSetTevDirect((GXTevStageID)s); }
PPC_NATIVE_OVERRIDE_VOID(80171b58, GX__SetTevDirect_80171b58, (uint32_t s), (s));
static void GX__SetTevDirect_80171b58_gx(uint32_t s) { GXSetTevDirect((GXTevStageID)s); }
GX_DEFERRED_OVERRIDE_VOID(80171b58, GX__SetTevDirect_80171b58, (uint32_t s), (s));
extern "C" void GX__SetTevIndWarp_80171ba0(uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms) {
static void GX__SetTevIndWarp_80171ba0_gx(uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms) {
GXSetTevIndWarp((GXTevStageID)std::min(ts, 15u), (GXIndTexStageID)std::min(is, 3u),
(GXBool)so, (GXBool)rm, (GXIndTexMtxID)ms);
}
PPC_NATIVE_OVERRIDE_VOID(80171ba0, GX__SetTevIndWarp_80171ba0, (uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms), (ts, is, so, rm, ms));
GX_DEFERRED_OVERRIDE_VOID(80171ba0, GX__SetTevIndWarp_80171ba0, (uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms), (ts, is, so, rm, ms));
extern "C" void GX__SetTevIndirect_801717ac(uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as) {
static void GX__SetTevIndirect_801717ac_gx(uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as) {
GXSetTevIndirect((GXTevStageID)std::min(ts,15u), (GXIndTexStageID)std::min(is,3u), (GXIndTexFormat)f,
(GXIndTexBiasSel)bs, (GXIndTexMtxID)ms, (GXIndTexWrap)ws, (GXIndTexWrap)wt,
(GXBool)ap, (GXBool)il, (GXIndTexAlphaSel)as);
}
PPC_NATIVE_OVERRIDE_VOID(801717ac, GX__SetTevIndirect_801717ac, (uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as), (ts, is, f, bs, ms, ws, wt, ap, il, as));
GX_DEFERRED_OVERRIDE_VOID(801717ac, GX__SetTevIndirect_801717ac, (uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as), (ts, is, f, bs, ms, ws, wt, ap, il, as));
+17 -13
View File
@@ -13,6 +13,9 @@ extern "C" void GX__InitFifoBase_8016c7c8(uint32_t fa, uint32_t ba, uint32_t s);
extern "C" void GX__SetGPFifo_8016cb2c(uint32_t fa);
extern "C" GXFifoObj* GXInit(void* base, u32 size);
static void GX__Init_gx(uint32_t fifoBase, uint32_t fifoSize) { GXInit(GuestToHostPtr(fifoBase, fifoSize), fifoSize); }
static void GX__Flush_gx() { GXFlush(); }
// ============================================================================
// GXInit
// ============================================================================
@@ -28,7 +31,7 @@ extern "C" uint32_t GX__Init_8016b850(uint32_t fifoBase, uint32_t fifoSize)
constexpr uint32_t kGXDataAddr = 0x803437C0u;
constexpr uint32_t kGXDataSize = 0x600u;
GXInit(GuestToHostPtr(fifoBase, fifoSize), fifoSize);
GxThread::Post(&GX__Init_gx, fifoBase, fifoSize);
// Initialize GXData structure in guest memory
try {
@@ -89,19 +92,20 @@ PPC_NATIVE_OVERRIDE(8016b850, GX__Init_8016b850, uint32_t, (uint32_t fifoBase, u
// FIFO Management
// ============================================================================
extern "C" void GX__InitFifoBase_8016c7c8(uint32_t fa, uint32_t ba, uint32_t s) { GXInitFifoBase((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)), GuestToHostPtr(ba, s), s); }
static void GX__InitFifoBase_8016c7c8_gx(uint32_t fa, uint32_t ba, uint32_t s) { GXInitFifoBase((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)), GuestToHostPtr(ba, s), s); }
extern "C" void GX__InitFifoBase_8016c7c8(uint32_t fa, uint32_t ba, uint32_t s) { GxThread::Post(&GX__InitFifoBase_8016c7c8_gx, fa, ba, s); }
extern "C" void GX__SetCPUFifo_8016c94c(uint32_t fa) { GXSetCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
PPC_NATIVE_OVERRIDE_VOID(8016c94c, GX__SetCPUFifo_8016c94c, (uint32_t fa), (fa));
static void GX__SetCPUFifo_8016c94c_gx(uint32_t fa) { GXSetCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
GX_DEFERRED_OVERRIDE_VOID(8016c94c, GX__SetCPUFifo_8016c94c, (uint32_t fa), (fa));
extern "C" void GX__SetGPFifo_8016cb2c(uint32_t fa) { GXSetGPFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
PPC_NATIVE_OVERRIDE_VOID(8016cb2c, GX__SetGPFifo_8016cb2c, (uint32_t fa), (fa));
static void GX__SetGPFifo_8016cb2c_gx(uint32_t fa) { GXSetGPFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
GX_DEFERRED_OVERRIDE_VOID(8016cb2c, GX__SetGPFifo_8016cb2c, (uint32_t fa), (fa));
extern "C" void __GX__SaveFifo_8016cdbc(uint32_t fa) { GXSaveCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
PPC_NATIVE_OVERRIDE_VOID(8016cdbc, __GX__SaveFifo_8016cdbc, (uint32_t fa), (fa));
static void __GX__SaveFifo_8016cdbc_gx(uint32_t fa) { GXSaveCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
GX_DEFERRED_OVERRIDE_VOID(8016cdbc, __GX__SaveFifo_8016cdbc, (uint32_t fa), (fa));
extern "C" void GX__GetCPUFifo_8016cf10(uint32_t fa) { auto* d=(GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)); auto* s=GXGetCPUFifo(); if(d&&s) std::memcpy(d,s,sizeof(GXFifoObj)); }
PPC_NATIVE_OVERRIDE_VOID(8016cf10, GX__GetCPUFifo_8016cf10, (uint32_t fa), (fa));
static void GX__GetCPUFifo_8016cf10_gx(uint32_t fa) { auto* d=(GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)); auto* s=GXGetCPUFifo(); if(d&&s) std::memcpy(d,s,sizeof(GXFifoObj)); }
GX_DEFERRED_OVERRIDE_VOID(8016cf10, GX__GetCPUFifo_8016cf10, (uint32_t fa), (fa));
extern "C" void __GX__FifoInit_8016d180()
{
@@ -181,7 +185,7 @@ extern "C" void GX__BeginDisplayList_80172e00(uint32_t la, uint32_t s) {
// Mirror the guest fifo-object fields the FIFO write path consumes so
// HleFifoWrite never has to read them back out of guest memory.
BeginDisplayListRecording(la, s);
GXFlush();
GxThread::Post(&GX__Flush_gx);
GX__GetCPUFifo_8016cf10(0x80344710);
GX__SetCPUFifo_8016c94c(0x80344090);
} catch (...) {}
@@ -190,7 +194,7 @@ PPC_NATIVE_OVERRIDE_VOID(80172e00, GX__BeginDisplayList_80172e00, (uint32_t la,
extern "C" uint32_t GX__EndDisplayList_80172eb4() {
try {
GXFlush();
GxThread::Post(&GX__Flush_gx);
GX__GetCPUFifo_8016cf10(0x80344090);
const uint8_t wrapped = Memory::Read8(kDlFifoAddr + kDlWrapFlagOffset);
GX__SetCPUFifo_8016c94c(0x80344710);
@@ -229,5 +233,5 @@ PPC_NATIVE_OVERRIDE(80172EB4, GX__EndDisplayList_80172eb4, uint32_t, (), ());
// Flush
// ============================================================================
extern "C" void GX__Flush_8016e654() { GXFlush(); }
extern "C" void GX__Flush_8016e654() { GxThread::Post(&GX__Flush_gx); }
PPC_NATIVE_OVERRIDE_VOID(8016e654, GX__Flush_8016e654, (), ());
+46 -9
View File
@@ -3,6 +3,7 @@
#include "hle_stubs.h"
#include "memory.h"
#include "gx_guest_write.h"
#include "gx_thread.h"
#include "ppc_runtime.h"
#include "aurora_events.h"
#include "gx_texture_binding_contract.h"
@@ -72,6 +73,7 @@ extern "C" uint32_t OS__GetCurrentThread_801a98b0_hle();
extern "C" void GX__SetCPUFifo_8016c94c(uint32_t fifoAddr);
extern "C" void GX__SetDirtyState_8016ee78();
extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes);
void GX__CallDisplayList_gx(uint32_t listAddr, uint32_t nbytes);
extern std::atomic_bool g_auroraFrameActive;
extern std::atomic_bool g_auroraFrameHadWork;
@@ -141,8 +143,6 @@ struct TexObjSlot : TexObjMeta {
// index vectors can carry slot pointers instead of re-looking-up keys.
uint32_t objAddr = 0;
// Aurora-side object. Created lazily by CreateHostTexObj.
std::unique_ptr<HleTexObj> host;
// Byte-exact shadow of the last-decoded guest GXTexObj; served without a
// diff only while guest bytes still match it. Games mutate these structs
@@ -343,18 +343,11 @@ TexObjMeta& GetTexObjMeta(uint32_t addr);
TexObjMeta ExtractTexObjMetaFromGuest(uint32_t addr);
bool TryGetOrExtractTexObjMeta(uint32_t addr, TexObjMeta& outMeta);
TlutObjMeta& GetTlutObjMeta(uint32_t addr);
GXTexObj* GetHostTexObj(uint32_t addr);
GXTexObj* TryGetHostTexObj(uint32_t addr);
GXTexObj* CreateHostTexObj(uint32_t addr);
void MarkHostTexObjConstructed(uint32_t addr);
void MarkTexObjsDirtyForRange(uint32_t addr, uint32_t size);
// Invalidates GPU-only GXCopyTex results when guest CPU writes are made visible
// over their destination. This is the allocation-reuse generation boundary
// for copy textures; it must be called for every data-cache store/flush range.
void InvalidateEfbCopyDestinationsForRange(uint32_t addr, uint32_t size);
GXTlutObj* CreateHostTlutObj(uint32_t addr);
void MarkHostTlutObjConstructed(uint32_t addr);
GXTlutObj* GetHostTlutObj(uint32_t addr);
void MarkTlutObjsDirtyForRange(uint32_t addr, uint32_t size);
// One-stop invalidation for DMA-class host-side writes into guest RAM (DVD
// reads, DCZeroRange, LC stores): texobjs + TLUTs + EFB copies + display
@@ -363,6 +356,50 @@ void MarkTlutObjsDirtyForRange(uint32_t addr, uint32_t size);
extern "C" void GxNotifyGuestRamDmaWrite(uint32_t addr, uint32_t size);
void HleFifoWrite(u32 val, uint32_t sizeBytes);
// GX-thread side of the write-gather pipe: parses a run of FIFO bytes.
extern "C" void GxFifoConsumeBytes(const uint8_t* data, uint32_t sizeBytes);
// ---------------------------------------------------------------------------
// GX thread split. GX_DEFERRED_OVERRIDE_VOID registers a guest-facing front
// that posts `name_gx` (the aurora work) to the GX thread; see gx_thread.h.
// ---------------------------------------------------------------------------
#define GX_COMMA_ARGS(...) , ##__VA_ARGS__
#define GX_DEFERRED_OVERRIDE_VOID(addr_hex, name, arg_list, call_list) \
extern "C" void name arg_list { GxThread::Post(&name##_gx GX_COMMA_ARGS call_list); } \
PPC_NATIVE_OVERRIDE_VOID(addr_hex, name, arg_list, call_list)
// Game-thread mirrors of the vertex descriptor state, kept for the GXGet*
// overrides; the GX thread owns g_hleGxState.
struct GxGameSideVertexState {
GXAttrType vtxDesc[26]{};
VtxAttrFmt vtxAttrFmt[8][26]{};
};
extern GxGameSideVertexState g_gxGameVertexState;
// Texture objects: the game thread keeps the meta table (guest shadows, dirty
// flags, getters); the GX thread keeps the aurora objects and rebuilds them
// from the meta snapshot each load carries.
struct GxTexObjLoad {
TexObjMeta meta;
uint32_t objAddr = 0;
uint32_t tid = 0;
bool upload = false;
};
void GxHostLoadTexObj_gx(GxTexObjLoad load);
void GxHostBindPlaceholder_gx(uint32_t tid);
struct GxTlutLoad {
TlutObjMeta meta;
uint32_t objAddr = 0;
uint32_t tlut = 0;
bool rebuild = false;
bool valid = false;
};
void GxHostLoadTlut_gx(GxTlutLoad load);
void GxHostDestroyCopyTex_gx(uint32_t copyAddr);
// Backs shared by the EGG helpers.
void GX__SetTevColor_gx(uint32_t id, uint32_t colorWord);
void GX__SetTevKColor_gx(uint32_t id, uint32_t colorWord);
void GX__Begin_gx(uint32_t t, uint32_t vf, uint32_t nv);
void SubmitAttribute(GXAttr attr, float* comps, const VtxAttrFmt& fmt, const u32* rawComps = nullptr);
void SubmitIndexedAttribute(GXAttr attr, uint32_t index);
+41 -17
View File
@@ -27,12 +27,29 @@ bool ValidateGuestLightObj(uint32_t addr, const char* operation) {
return false;
}
float ReadGuestLightFloat(uint32_t addr, uint32_t offset) {
return Memory::ReadFloat32(addr + offset);
// The SDK copies the light object into the FIFO at call time, so the guest
// object is snapshotted on the game thread and decoded on the GX thread.
struct GuestLightSnapshot {
uint32_t words[kLightObjSize / 4];
};
GuestLightSnapshot SnapshotGuestLight(uint32_t addr) {
GuestLightSnapshot snapshot{};
for (uint32_t i = 0; i < kLightObjSize / 4; ++i) {
snapshot.words[i] = Memory::Read32(addr + i * 4);
}
return snapshot;
}
void InitializeHostLightFromGuest(GXLightObj& host, uint32_t guestAddr) {
GXInitLightColor(&host, DecodeGxColor(Memory::Read32(guestAddr + kColorOffset)));
float ReadGuestLightFloat(const GuestLightSnapshot& light, uint32_t offset) {
float value;
const uint32_t bits = light.words[offset / 4];
std::memcpy(&value, &bits, sizeof(value));
return value;
}
void InitializeHostLightFromGuest(GXLightObj& host, const GuestLightSnapshot& guestAddr) {
GXInitLightColor(&host, DecodeGxColor(guestAddr.words[kColorOffset / 4]));
GXInitLightAttn(
&host,
ReadGuestLightFloat(guestAddr, kAttnAOffset + 0),
@@ -62,11 +79,14 @@ void InitializeHostLightFromGuest(GXLightObj& host, uint32_t guestAddr) {
// Light Object Load
// ============================================================================
static void GX__LoadLightObjImm_gx(GuestLightSnapshot light, uint32_t lid) {
GXLightObj host{};
InitializeHostLightFromGuest(host, light);
GXLoadLightObjImm(&host, static_cast<GXLightID>(lid));
}
extern "C" void GX__LoadLightObjImm_80170320(uint32_t la, uint32_t lid) {
if (!ValidateGuestLightObj(la, "GXLoadLightObjImm")) return;
GXLightObj host{};
InitializeHostLightFromGuest(host, la);
GXLoadLightObjImm(&host, static_cast<GXLightID>(lid));
GxThread::Post(&GX__LoadLightObjImm_gx, SnapshotGuestLight(la), lid);
}
PPC_NATIVE_OVERRIDE_VOID(80170320, GX__LoadLightObjImm_80170320, (uint32_t la, uint32_t lid), (la, lid));
@@ -74,24 +94,28 @@ PPC_NATIVE_OVERRIDE_VOID(80170320, GX__LoadLightObjImm_80170320, (uint32_t la, u
// Channel Control
// ============================================================================
extern "C" void GX__SetChanAmbColor_8017039c(uint32_t c, uint32_t cp) {
static void GX__SetChanAmbColor_gx(uint32_t c, uint32_t colorWord) {
EnsureAuroraFrameActive();
GXColor color = DecodeGxColor(Memory::Read32(cp));
GXSetChanAmbColor((GXChannelID)c, color);
GXSetChanAmbColor((GXChannelID)c, DecodeGxColor(colorWord));
}
extern "C" void GX__SetChanAmbColor_8017039c(uint32_t c, uint32_t cp) {
GxThread::Post(&GX__SetChanAmbColor_gx, c, Memory::Read32(cp));
}
PPC_NATIVE_OVERRIDE_VOID(8017039c, GX__SetChanAmbColor_8017039c, (uint32_t c, uint32_t cp), (c, cp));
extern "C" void GX__SetChanMatColor_80170474(uint32_t c, uint32_t cp) {
static void GX__SetChanMatColor_gx(uint32_t c, uint32_t colorWord) {
EnsureAuroraFrameActive();
GXColor color = DecodeGxColor(Memory::Read32(cp));
GXSetChanMatColor((GXChannelID)c, color);
GXSetChanMatColor((GXChannelID)c, DecodeGxColor(colorWord));
}
extern "C" void GX__SetChanMatColor_80170474(uint32_t c, uint32_t cp) {
GxThread::Post(&GX__SetChanMatColor_gx, c, Memory::Read32(cp));
}
PPC_NATIVE_OVERRIDE_VOID(80170474, GX__SetChanMatColor_80170474, (uint32_t c, uint32_t cp), (c, cp));
extern "C" void GX__SetNumChans_8017054c(uint32_t n) { GXSetNumChans((u8)n); }
PPC_NATIVE_OVERRIDE_VOID(8017054c, GX__SetNumChans_8017054c, (uint32_t n), (n));
static void GX__SetNumChans_8017054c_gx(uint32_t n) { GXSetNumChans((u8)n); }
GX_DEFERRED_OVERRIDE_VOID(8017054c, GX__SetNumChans_8017054c, (uint32_t n), (n));
extern "C" void GX__SetChanCtrl_80170570(uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af) {
static void GX__SetChanCtrl_80170570_gx(uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af) {
GXSetChanCtrl((GXChannelID)ch, en!=0, (GXColorSrc)as, (GXColorSrc)ms, lm, (GXDiffuseFn)df, (GXAttnFn)af);
}
PPC_NATIVE_OVERRIDE_VOID(80170570, GX__SetChanCtrl_80170570, (uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af), (ch, en, as, ms, lm, df, af));
GX_DEFERRED_OVERRIDE_VOID(80170570, GX__SetChanCtrl_80170570, (uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af), (ch, en, as, ms, lm, df, af));
+1 -119
View File
@@ -15,48 +15,9 @@ using TextureHandle = std::shared_ptr<TextureRef>;
// HleTexObj now lives in gx_internal.h: it is embedded in TexObjSlot so the
// metadata and the Aurora object share one hash-map entry.
struct HostGXTlutObj {
alignas(GXTlutObj) std::byte publicStorage[sizeof(GXTlutObj)]{};
aurora::gfx::TextureHandle ref;
};
struct HleTlutObj {
static constexpr size_t kTlutObjStorageSize =
(sizeof(GXTlutObj) >= sizeof(HostGXTlutObj)) ? sizeof(GXTlutObj) : sizeof(HostGXTlutObj);
static constexpr size_t kTlutObjStorageAlign =
(alignof(GXTlutObj) >= alignof(HostGXTlutObj)) ? alignof(GXTlutObj) : alignof(HostGXTlutObj);
using Storage = std::aligned_storage_t<kTlutObjStorageSize, kTlutObjStorageAlign>;
Storage storage{};
bool storageLive = false;
bool constructed = false;
HleTlutObj() = default;
~HleTlutObj() { Destroy(); }
HostGXTlutObj* HostObj() { return reinterpret_cast<HostGXTlutObj*>(&storage); }
GXTlutObj* PublicPtr() { return reinterpret_cast<GXTlutObj*>(HostObj()->publicStorage); }
void EnsureStorageLive() {
if (!storageLive) {
new (&storage) HostGXTlutObj();
storageLive = true;
}
}
void Destroy() {
if (storageLive) {
GXDestroyTlutObj(PublicPtr());
std::destroy_at(HostObj());
std::memset(&storage, 0, sizeof(storage));
storageLive = false;
constructed = false;
}
}
};
// The aurora TLUT objects live on the GX thread (gx_texture.cpp).
std::mutex g_texObjMutex;
std::map<uint32_t, std::unique_ptr<HleTlutObj>> g_HostTlutObjMap;
std::mutex g_tlutObjMutex;
// The texobj table. One slot per guest GXTexObj address holds both the decoded
@@ -548,52 +509,6 @@ TlutObjMeta& GetTlutObjMeta(uint32_t addr) {
return g_TlutObjMeta[addr];
}
GXTexObj* GetHostTexObj(uint32_t addr) {
TexObjSlot* slot = FindTexObjSlot(addr);
if (slot == nullptr || !slot->host || !slot->host->constructed) {
const CpuContext* cpu = TryGetCpuContext();
RT_LOGF(RT_TAG_GX,
"GXTex: invalid GXTexObj @0x%08X (PC=0x%08X, LR=0x%08X, CTR=0x%08X)\n",
addr, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u, cpu ? cpu->ctr : 0u);
std::fflush(stderr);
GXTexObj* obj = CreateHostTexObj(addr);
MarkHostTexObjConstructed(addr);
return obj;
}
return slot->host->PublicPtr();
}
GXTexObj* CreateHostTexObj(uint32_t addr) {
TexObjSlot& slot = FindOrCreateTexObjSlot(addr);
// GXInitTexObj/GXInitTexObjCI land here and rewrite the entire guest struct,
// resetting the LOD/filter words the HLE does not mirror into this cache.
// The shadow captured before that rewrite mismatches afterwards, so the
// next lookup takes exactly one guest re-decode and picks those defaults
// (and the computed mipmap maxLod) up.
if (!slot.host) {
slot.host = std::make_unique<HleTexObj>();
} else {
slot.host->Destroy();
}
slot.host->EnsureStorageLive();
return slot.host->PublicPtr();
}
void MarkHostTexObjConstructed(uint32_t addr) {
TexObjSlot* slot = FindTexObjSlot(addr);
if (slot != nullptr && slot->host) {
slot->host->constructed = true;
}
}
GXTexObj* TryGetHostTexObj(uint32_t addr) {
TexObjSlot* slot = FindTexObjSlot(addr);
if (slot != nullptr && slot->host && slot->host->constructed) {
return slot->host->PublicPtr();
}
return nullptr;
}
void MarkTexObjsDirtyForRange(uint32_t addr, uint32_t size) {
if (size == 0) {
return;
@@ -656,39 +571,6 @@ extern "C" void GxNotifyGuestRamDmaWrite(uint32_t addr, uint32_t size) {
GxNotifyDisplayListMemoryWrite(addr, size);
}
GXTlutObj* CreateHostTlutObj(uint32_t addr) {
auto& slot = g_HostTlutObjMap[addr];
if (!slot) {
slot = std::make_unique<HleTlutObj>();
} else {
slot->Destroy();
}
slot->EnsureStorageLive();
return slot->PublicPtr();
}
void MarkHostTlutObjConstructed(uint32_t addr) {
auto it = g_HostTlutObjMap.find(addr);
if (it != g_HostTlutObjMap.end() && it->second) {
it->second->constructed = true;
}
}
GXTlutObj* GetHostTlutObj(uint32_t addr) {
auto it = g_HostTlutObjMap.find(addr);
if (it == g_HostTlutObjMap.end() || !it->second || !it->second->constructed) {
const CpuContext* cpu = TryGetCpuContext();
RT_LOGF(RT_TAG_GX,
"GXTex: invalid GXTlutObj @0x%08X (PC=0x%08X, LR=0x%08X, CTR=0x%08X)\n",
addr, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u, cpu ? cpu->ctr : 0u);
std::fflush(stderr);
GXTlutObj* obj = CreateHostTlutObj(addr);
MarkHostTlutObjConstructed(addr);
return obj;
}
return it->second->PublicPtr();
}
void MarkTlutObjsDirtyForRange(uint32_t addr, uint32_t size) {
if (size == 0) {
return;
+31 -27
View File
@@ -5,84 +5,88 @@
// Blend Mode
// ============================================================================
extern "C" void GX__SetBlendMode_8017277c(uint32_t t, uint32_t s, uint32_t d, uint32_t op) {
static void GX__SetBlendMode_8017277c_gx(uint32_t t, uint32_t s, uint32_t d, uint32_t op) {
GXSetBlendMode(static_cast<GXBlendMode>(t), static_cast<GXBlendFactor>(s),
static_cast<GXBlendFactor>(d), static_cast<GXLogicOp>(op));
}
PPC_NATIVE_OVERRIDE_VOID(8017277c, GX__SetBlendMode_8017277c, (uint32_t t, uint32_t s, uint32_t d, uint32_t op), (t, s, d, op));
GX_DEFERRED_OVERRIDE_VOID(8017277c, GX__SetBlendMode_8017277c, (uint32_t t, uint32_t s, uint32_t d, uint32_t op), (t, s, d, op));
extern "C" void GX__SetColorUpdate_801727cc(uint32_t en) {
static void GX__SetColorUpdate_801727cc_gx(uint32_t en) {
GXSetColorUpdate(static_cast<GXBool>(en));
}
PPC_NATIVE_OVERRIDE_VOID(801727cc, GX__SetColorUpdate_801727cc, (uint32_t en), (en));
GX_DEFERRED_OVERRIDE_VOID(801727cc, GX__SetColorUpdate_801727cc, (uint32_t en), (en));
extern "C" void GX__SetAlphaUpdate_801727f8(uint32_t en) {
static void GX__SetAlphaUpdate_801727f8_gx(uint32_t en) {
GXSetAlphaUpdate(static_cast<GXBool>(en));
}
PPC_NATIVE_OVERRIDE_VOID(801727f8, GX__SetAlphaUpdate_801727f8, (uint32_t en), (en));
GX_DEFERRED_OVERRIDE_VOID(801727f8, GX__SetAlphaUpdate_801727f8, (uint32_t en), (en));
// ============================================================================
// Z Buffer
// ============================================================================
extern "C" void GX__SetZMode_80172824(uint32_t ce, uint32_t f, uint32_t ue) {
static void GX__SetZMode_80172824_gx(uint32_t ce, uint32_t f, uint32_t ue) {
GXSetZMode(static_cast<GXBool>(ce), static_cast<GXCompare>(f), static_cast<GXBool>(ue));
}
PPC_NATIVE_OVERRIDE_VOID(80172824, GX__SetZMode_80172824, (uint32_t ce, uint32_t f, uint32_t ue), (ce, f, ue));
GX_DEFERRED_OVERRIDE_VOID(80172824, GX__SetZMode_80172824, (uint32_t ce, uint32_t f, uint32_t ue), (ce, f, ue));
extern "C" void GX__SetZCompLoc_80172858(uint32_t bt) {
static void GX__SetZCompLoc_80172858_gx(uint32_t bt) {
GXSetZCompLoc(static_cast<GXBool>(bt));
}
PPC_NATIVE_OVERRIDE_VOID(80172858, GX__SetZCompLoc_80172858, (uint32_t bt), (bt));
GX_DEFERRED_OVERRIDE_VOID(80172858, GX__SetZCompLoc_80172858, (uint32_t bt), (bt));
// ============================================================================
// Pixel Format and Dither
// ============================================================================
extern "C" void GX__SetPixelFmt_80172888(uint32_t pf, uint32_t zf) {
static void GX__SetPixelFmt_80172888_gx(uint32_t pf, uint32_t zf) {
GXSetPixelFmt(static_cast<GXPixelFmt>(pf), static_cast<GXZFmt16>(zf));
}
PPC_NATIVE_OVERRIDE_VOID(80172888, GX__SetPixelFmt_80172888, (uint32_t pf, uint32_t zf), (pf, zf));
GX_DEFERRED_OVERRIDE_VOID(80172888, GX__SetPixelFmt_80172888, (uint32_t pf, uint32_t zf), (pf, zf));
extern "C" void GX__SetDither_80172930(uint32_t d) {
static void GX__SetDither_80172930_gx(uint32_t d) {
GXSetDither(static_cast<GXBool>(d));
}
PPC_NATIVE_OVERRIDE_VOID(80172930, GX__SetDither_80172930, (uint32_t d), (d));
GX_DEFERRED_OVERRIDE_VOID(80172930, GX__SetDither_80172930, (uint32_t d), (d));
extern "C" void GX__SetDstAlpha_8017295c(uint32_t en, uint32_t a) {
static void GX__SetDstAlpha_8017295c_gx(uint32_t en, uint32_t a) {
GXSetDstAlpha(static_cast<GXBool>(en), static_cast<u8>(a));
}
PPC_NATIVE_OVERRIDE_VOID(8017295c, GX__SetDstAlpha_8017295c, (uint32_t en, uint32_t a), (en, a));
GX_DEFERRED_OVERRIDE_VOID(8017295c, GX__SetDstAlpha_8017295c, (uint32_t en, uint32_t a), (en, a));
// ============================================================================
// Fog and Alpha Compare
// ============================================================================
static void GX__SetFog_gx(uint32_t t, float sz, float ez, float nz, float fz, uint32_t colorWord) {
GXSetFog((GXFogType)t, sz, ez, nz, fz, DecodeGxColor(colorWord));
}
extern "C" void GX__SetFog_801722cc(uint32_t t, float sz, float ez, float nz, float fz, uint32_t cp) {
GXSetFog((GXFogType)t, sz, ez, nz, fz, DecodeGxColor(Memory::Read32(cp)));
// The colour is copied out of guest memory at call time, as the SDK does.
GxThread::Post(&GX__SetFog_gx, t, sz, ez, nz, fz, Memory::Read32(cp));
}
PPC_NATIVE_OVERRIDE_VOID(801722cc, GX__SetFog_801722cc, (uint32_t t, float sz, float ez, float nz, float fz, uint32_t cp), (t, sz, ez, nz, fz, cp));
extern "C" void GX__SetAlphaCompare_80172088(uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1) {
static void GX__SetAlphaCompare_80172088_gx(uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1) {
g_alphaCompareValid = true;
GXSetAlphaCompare((GXCompare)c0, (u8)r0, (GXAlphaOp)op, (GXCompare)c1, (u8)r1);
}
PPC_NATIVE_OVERRIDE_VOID(80172088, GX__SetAlphaCompare_80172088, (uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1), (c0, r0, op, c1, r1));
GX_DEFERRED_OVERRIDE_VOID(80172088, GX__SetAlphaCompare_80172088, (uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1), (c0, r0, op, c1, r1));
extern "C" void GX__SetZTexture_801720c0(uint32_t op, uint32_t f, uint32_t b) { GXSetZTexture((GXZTexOp)op, (GXTexFmt)f, b); }
PPC_NATIVE_OVERRIDE_VOID(801720c0, GX__SetZTexture_801720c0, (uint32_t op, uint32_t f, uint32_t b), (op, f, b));
static void GX__SetZTexture_801720c0_gx(uint32_t op, uint32_t f, uint32_t b) { GXSetZTexture((GXZTexOp)op, (GXTexFmt)f, b); }
GX_DEFERRED_OVERRIDE_VOID(801720c0, GX__SetZTexture_801720c0, (uint32_t op, uint32_t f, uint32_t b), (op, f, b));
// ============================================================================
// Culling and Clipping
// ============================================================================
extern "C" void GX__SetCullMode_8016f3b8(uint32_t m) {
static void GX__SetCullMode_8016f3b8_gx(uint32_t m) {
GXSetCullMode(static_cast<GXCullMode>(m));
}
PPC_NATIVE_OVERRIDE_VOID(8016f3b8, GX__SetCullMode_8016f3b8, (uint32_t m), (m));
GX_DEFERRED_OVERRIDE_VOID(8016f3b8, GX__SetCullMode_8016f3b8, (uint32_t m), (m));
extern "C" void GX__SetCoPlanar_8016f3e0(uint32_t en) { GXSetCoPlanar((GXBool)en); }
PPC_NATIVE_OVERRIDE_VOID(8016f3e0, GX__SetCoPlanar_8016f3e0, (uint32_t en), (en));
static void GX__SetCoPlanar_8016f3e0_gx(uint32_t en) { GXSetCoPlanar((GXBool)en); }
GX_DEFERRED_OVERRIDE_VOID(8016f3e0, GX__SetCoPlanar_8016f3e0, (uint32_t en), (en));
extern "C" void GX__SetClipMode_8017351c(uint32_t m) { GXSetClipMode((GXClipMode)m); }
PPC_NATIVE_OVERRIDE_VOID(8017351c, GX__SetClipMode_8017351c, (uint32_t m), (m));
static void GX__SetClipMode_8017351c_gx(uint32_t m) { GXSetClipMode((GXClipMode)m); }
GX_DEFERRED_OVERRIDE_VOID(8017351c, GX__SetClipMode_8017351c, (uint32_t m), (m));
+17 -18
View File
@@ -3,18 +3,7 @@
extern "C" void __GXSetSUTexRegs();
// ============================================================================
// FIFO Write Helpers
// ============================================================================
extern "C" void GX_HLE_FIFO_WriteFloat(float val) {
u32 raw; std::memcpy(&raw, &val, 4);
try { HleFifoWrite(raw, 4); } catch (...) { RT_LOGF(RT_TAG_GX, "FIFO write float failed\n"); }
}
extern "C" void GX_HLE_FIFO_Write32(uint32_t val) { HleFifoWrite(val, 4); }
extern "C" void GX_HLE_FIFO_Write16(uint16_t val) { HleFifoWrite(static_cast<u32>(val), 2); }
extern "C" void GX_HLE_FIFO_Write8(uint8_t val) { HleFifoWrite(static_cast<u32>(val), 1); }
// The FIFO write helpers (GX_HLE_FIFO_Write*) live in gx_fifo.cpp.
extern "C" void GX__SetDrawSync_8016ed08(uint32_t token) {
(void)token;
@@ -36,15 +25,21 @@ extern "C" void GX__FinishInterruptHandler_8016ed94() {
}
PPC_NATIVE_OVERRIDE_VOID(8016ed94, GX__FinishInterruptHandler_8016ed94, (), ());
static void GX__DrawDone_gx() { GXDrawDone(); }
extern "C" void GX__DrawDone_8016eab0() {
try { Memory::Write8(kGxDrawDoneFlagAddr, 0); } catch (...) {}
GXDrawDone(); GX__FinishInterruptHandler_8016ed94();
// Hardware blocks here until the GP has consumed the FIFO: post the drain
// and wait for the GX thread to reach it before raising the finish flags.
GxThread::Post(&GX__DrawDone_gx);
GxThread::Drain();
GX__FinishInterruptHandler_8016ed94();
}
PPC_NATIVE_OVERRIDE_VOID(8016eab0, GX__DrawDone_8016eab0, (), ());
static void GX__PixModeSync_gx() { GXPixModeSync(); }
extern "C" void GX__PixModeSync_8016eb70() {
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {}
GXPixModeSync();
GxThread::Post(&GX__PixModeSync_gx);
}
PPC_NATIVE_OVERRIDE_VOID(8016eb70, GX__PixModeSync_8016eb70, (), ());
@@ -59,8 +54,9 @@ PPC_NATIVE_OVERRIDE_VOID(8016b720, __GX__InitRevisionBits_8016b720, (), ());
// Texture State Management - Aurora handles internally
// ============================================================================
static void __GX__SetSUTexRegs_gx() { __GXSetSUTexRegs(); }
extern "C" void __GX__SetSUTexRegs_801712f0() {
__GXSetSUTexRegs();
GxThread::Post(&__GX__SetSUTexRegs_gx);
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {}
}
PPC_NATIVE_OVERRIDE_VOID(801712f0, __GX__SetSUTexRegs_801712f0, (), ());
@@ -81,8 +77,9 @@ PPC_NATIVE_OVERRIDE_VOID(80171c28, __GX__FlushTextureState_80171c28, (), ());
// Copy Configuration - No-ops for features Aurora doesn't use
// ============================================================================
static void GX__SetDispCopyFrame2Field_gx(uint32_t f) { GXSetDispCopyFrame2Field(f); }
extern "C" void GX__SetDispCopyFrame2Field_8016f5f8(uint32_t f) {
GXSetDispCopyFrame2Field(f);
GxThread::Post(&GX__SetDispCopyFrame2Field_gx, f);
try {
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
if (gd) {
@@ -93,8 +90,9 @@ extern "C" void GX__SetDispCopyFrame2Field_8016f5f8(uint32_t f) {
}
PPC_NATIVE_OVERRIDE_VOID(8016f5f8, GX__SetDispCopyFrame2Field_8016f5f8, (uint32_t f), (f));
static void GX__SetCopyClamp_gx(uint32_t c) { GXSetCopyClamp(static_cast<GXFBClamp>(c)); }
extern "C" void GX__SetCopyClamp_8016f618(uint32_t c) {
GXSetCopyClamp(static_cast<GXFBClamp>(c));
GxThread::Post(&GX__SetCopyClamp_gx, c);
try {
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
if (gd) {
@@ -106,8 +104,9 @@ extern "C" void GX__SetCopyClamp_8016f618(uint32_t c) {
}
PPC_NATIVE_OVERRIDE_VOID(8016f618, GX__SetCopyClamp_8016f618, (uint32_t c), (c));
static void GX__ClearBoundingBox_gx() { GXClearBoundingBox(); }
extern "C" void GX__ClearBoundingBox_8016fecc() {
GXClearBoundingBox();
GxThread::Post(&GX__ClearBoundingBox_gx);
try {
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
if (gd) Memory::Write16(gd + 2, 0);
+36 -29
View File
@@ -24,7 +24,7 @@ inline bool TevSwapOk(uint32_t id) { return GxTevIdOk(id, GX_MAX_TEVSWAP, "TEV s
// TEV Stage Count and Order
// ============================================================================
extern "C" void GX__SetNumTevStages_801722a8(uint32_t n) {
static void GX__SetNumTevStages_801722a8_gx(uint32_t n) {
// GXSetNumTevStages takes a count, not an index, so the inclusive bound is
// GX_MAX_TEVSTAGE itself.
if (n > GX_MAX_TEVSTAGE) {
@@ -33,96 +33,103 @@ extern "C" void GX__SetNumTevStages_801722a8(uint32_t n) {
}
GXSetNumTevStages((u8)n);
}
PPC_NATIVE_OVERRIDE_VOID(801722a8, GX__SetNumTevStages_801722a8, (uint32_t n), (n));
GX_DEFERRED_OVERRIDE_VOID(801722a8, GX__SetNumTevStages_801722a8, (uint32_t n), (n));
extern "C" void GX__SetTevOp_80171c4c(uint32_t s, uint32_t m) {
static void GX__SetTevOp_80171c4c_gx(uint32_t s, uint32_t m) {
if (!TevStageOk(s)) return;
GXSetTevOp((GXTevStageID)s, (GXTevMode)m);
}
PPC_NATIVE_OVERRIDE_VOID(80171c4c, GX__SetTevOp_80171c4c, (uint32_t s, uint32_t m), (s, m));
GX_DEFERRED_OVERRIDE_VOID(80171c4c, GX__SetTevOp_80171c4c, (uint32_t s, uint32_t m), (s, m));
extern "C" void GX__SetTevOrder_8017214c(uint32_t s, uint32_t c, uint32_t m, uint32_t col) {
static void GX__SetTevOrder_8017214c_gx(uint32_t s, uint32_t c, uint32_t m, uint32_t col) {
if (!TevStageOk(s)) return;
GXSetTevOrder((GXTevStageID)s, (GXTexCoordID)c, (GXTexMapID)m, (GXChannelID)col);
}
PPC_NATIVE_OVERRIDE_VOID(8017214c, GX__SetTevOrder_8017214c, (uint32_t s, uint32_t c, uint32_t m, uint32_t col), (s, c, m, col));
GX_DEFERRED_OVERRIDE_VOID(8017214c, GX__SetTevOrder_8017214c, (uint32_t s, uint32_t c, uint32_t m, uint32_t col), (s, c, m, col));
// ============================================================================
// TEV Color/Alpha Inputs
// ============================================================================
extern "C" void GX__SetTevColorIn_80171ce0(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
static void GX__SetTevColorIn_80171ce0_gx(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
if (!TevStageOk(s)) return;
GXSetTevColorIn((GXTevStageID)s, (GXTevColorArg)a, (GXTevColorArg)b, (GXTevColorArg)c, (GXTevColorArg)d);
}
PPC_NATIVE_OVERRIDE_VOID(80171ce0, GX__SetTevColorIn_80171ce0, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
GX_DEFERRED_OVERRIDE_VOID(80171ce0, GX__SetTevColorIn_80171ce0, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
extern "C" void GX__SetTevAlphaIn_80171d20(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
static void GX__SetTevAlphaIn_80171d20_gx(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
if (!TevStageOk(s)) return;
GXSetTevAlphaIn((GXTevStageID)s, (GXTevAlphaArg)a, (GXTevAlphaArg)b, (GXTevAlphaArg)c, (GXTevAlphaArg)d);
}
PPC_NATIVE_OVERRIDE_VOID(80171d20, GX__SetTevAlphaIn_80171d20, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
GX_DEFERRED_OVERRIDE_VOID(80171d20, GX__SetTevAlphaIn_80171d20, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
// ============================================================================
// TEV Color/Alpha Operations
// ============================================================================
extern "C" void GX__SetTevColorOp_80171d60(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
static void GX__SetTevColorOp_80171d60_gx(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
if (!TevStageOk(s) || !TevRegOk(or_)) return;
GXSetTevColorOp((GXTevStageID)s, (GXTevOp)op, (GXTevBias)b, (GXTevScale)sc, (GXBool)cl, (GXTevRegID)or_);
}
PPC_NATIVE_OVERRIDE_VOID(80171d60, GX__SetTevColorOp_80171d60, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
GX_DEFERRED_OVERRIDE_VOID(80171d60, GX__SetTevColorOp_80171d60, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
extern "C" void GX__SetTevAlphaOp_80171db8(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
static void GX__SetTevAlphaOp_80171db8_gx(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
if (!TevStageOk(s) || !TevRegOk(or_)) return;
GXSetTevAlphaOp((GXTevStageID)s, (GXTevOp)op, (GXTevBias)b, (GXTevScale)sc, (GXBool)cl, (GXTevRegID)or_);
}
PPC_NATIVE_OVERRIDE_VOID(80171db8, GX__SetTevAlphaOp_80171db8, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
GX_DEFERRED_OVERRIDE_VOID(80171db8, GX__SetTevAlphaOp_80171db8, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
// ============================================================================
// TEV Color Registers
// ============================================================================
void GX__SetTevColor_gx(uint32_t id, uint32_t colorWord) {
GXSetTevColor((GXTevRegID)id, DecodeGxColor(colorWord));
}
extern "C" void GX__SetTevColor_80171e10(uint32_t id, uint32_t cp) {
if (!TevRegOk(id)) return;
const uint8_t* p=Memory::GetPointer(cp, 4); GXColor c; c.r=p[0]; c.g=p[1]; c.b=p[2]; c.a=p[3];
GXSetTevColor((GXTevRegID)id, c);
GxThread::Post(&GX__SetTevColor_gx, id, Memory::Read32(cp));
}
PPC_NATIVE_OVERRIDE_VOID(80171e10, GX__SetTevColor_80171e10, (uint32_t id, uint32_t cp), (id, cp));
static void GX__SetTevColorS10_gx(uint32_t id, uint32_t rg, uint32_t ba) {
GXColorS10 c;
c.r=static_cast<s16>(rg>>16); c.g=static_cast<s16>(rg&0xFFFFu); c.b=static_cast<s16>(ba>>16); c.a=static_cast<s16>(ba&0xFFFFu);
GXSetTevColorS10((GXTevRegID)id, c);
}
extern "C" void GX__SetTevColorS10_80171e70(uint32_t id, uint32_t cp) {
if (!TevRegOk(id)) return;
const uint8_t* p=Memory::GetPointer(cp, 8); GXColorS10 c;
c.r=(p[0]<<8)|p[1]; c.g=(p[2]<<8)|p[3]; c.b=(p[4]<<8)|p[5]; c.a=(p[6]<<8)|p[7];
GXSetTevColorS10((GXTevRegID)id, c);
GxThread::Post(&GX__SetTevColorS10_gx, id, Memory::Read32(cp), Memory::Read32(cp + 4));
}
PPC_NATIVE_OVERRIDE_VOID(80171e70, GX__SetTevColorS10_80171e70, (uint32_t id, uint32_t cp), (id, cp));
void GX__SetTevKColor_gx(uint32_t id, uint32_t colorWord) {
GXSetTevKColor((GXTevKColorID)id, DecodeGxColor(colorWord));
}
extern "C" void GX__SetTevKColor_80171ed4(uint32_t id, uint32_t cp) {
if (!TevKColorOk(id)) return;
const uint8_t* p=Memory::GetPointer(cp, 4); GXColor c; c.r=p[0]; c.g=p[1]; c.b=p[2]; c.a=p[3];
GXSetTevKColor((GXTevKColorID)id, c);
GxThread::Post(&GX__SetTevKColor_gx, id, Memory::Read32(cp));
}
PPC_NATIVE_OVERRIDE_VOID(80171ed4, GX__SetTevKColor_80171ed4, (uint32_t id, uint32_t cp), (id, cp));
extern "C" void GX__SetTevKColorSel_80171f30(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKColorSel((GXTevStageID)s, (GXTevKColorSel)sel); }
PPC_NATIVE_OVERRIDE_VOID(80171f30, GX__SetTevKColorSel_80171f30, (uint32_t s, uint32_t sel), (s, sel));
static void GX__SetTevKColorSel_80171f30_gx(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKColorSel((GXTevStageID)s, (GXTevKColorSel)sel); }
GX_DEFERRED_OVERRIDE_VOID(80171f30, GX__SetTevKColorSel_80171f30, (uint32_t s, uint32_t sel), (s, sel));
extern "C" void GX__SetTevKAlphaSel_80171f80(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKAlphaSel((GXTevStageID)s, (GXTevKAlphaSel)sel); }
PPC_NATIVE_OVERRIDE_VOID(80171f80, GX__SetTevKAlphaSel_80171f80, (uint32_t s, uint32_t sel), (s, sel));
static void GX__SetTevKAlphaSel_80171f80_gx(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKAlphaSel((GXTevStageID)s, (GXTevKAlphaSel)sel); }
GX_DEFERRED_OVERRIDE_VOID(80171f80, GX__SetTevKAlphaSel_80171f80, (uint32_t s, uint32_t sel), (s, sel));
// ============================================================================
// TEV Swap Tables
// ============================================================================
extern "C" void GX__SetTevSwapModeTable_8017200c(uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a) {
static void GX__SetTevSwapModeTable_8017200c_gx(uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a) {
if (!TevSwapOk(id)) return;
GXSetTevSwapModeTable((GXTevSwapSel)id, (GXTevColorChan)r, (GXTevColorChan)g, (GXTevColorChan)b, (GXTevColorChan)a);
}
PPC_NATIVE_OVERRIDE_VOID(8017200c, GX__SetTevSwapModeTable_8017200c, (uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a), (id, r, g, b, a));
GX_DEFERRED_OVERRIDE_VOID(8017200c, GX__SetTevSwapModeTable_8017200c, (uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a), (id, r, g, b, a));
extern "C" void GX__SetTevSwapMode_80171fd0(uint32_t s, uint32_t rs, uint32_t ts) {
static void GX__SetTevSwapMode_80171fd0_gx(uint32_t s, uint32_t rs, uint32_t ts) {
if (!TevStageOk(s) || !TevSwapOk(rs) || !TevSwapOk(ts)) return;
GXSetTevSwapMode((GXTevStageID)s, (GXTevSwapSel)rs, (GXTevSwapSel)ts);
}
PPC_NATIVE_OVERRIDE_VOID(80171fd0, GX__SetTevSwapMode_80171fd0, (uint32_t s, uint32_t rs, uint32_t ts), (s, rs, ts));
GX_DEFERRED_OVERRIDE_VOID(80171fd0, GX__SetTevSwapMode_80171fd0, (uint32_t s, uint32_t rs, uint32_t ts), (s, rs, ts));
+192 -84
View File
@@ -250,9 +250,8 @@ extern "C" void GX__InitTexObj_801707f8(uint32_t oa, uint32_t da, uint32_t w, ui
oa, da, w, h, f, ws, wt, m, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u);
}
const uint32_t canonicalDataAddr = CanonicalizeGxMainRamAddress(da);
std::lock_guard<std::mutex> guard(g_texObjMutex); GXTexObj* obj = CreateHostTexObj(oa); TexObjMeta& meta = GetTexObjMeta(oa);
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.userData=0; meta.needsUpload=true;
GXInitTexObj(obj, GuestToHostPtr(da), (u16)w, (u16)h, (GXTexFmt)f, (GXTexWrapMode)ws, (GXTexWrapMode)wt, (GXBool)m); MarkHostTexObjConstructed(oa);
// Also write to guest memory so reads work
WriteGuestTexObj(oa, canonicalDataAddr, (u16)w, (u16)h, f, ws, wt, m != 0, false, 0);
}
@@ -291,9 +290,8 @@ extern "C" void GX__InitTexObjCI_80170a04(uint32_t oa, uint32_t da, uint32_t w,
oa, da, w, h, f, ws, wt, m, tl, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u);
}
const uint32_t canonicalDataAddr = CanonicalizeGxMainRamAddress(da);
std::lock_guard<std::mutex> guard(g_texObjMutex); GXTexObj* obj = CreateHostTexObj(oa); TexObjMeta& meta = GetTexObjMeta(oa);
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.tlut=tl; meta.userData=0; meta.needsUpload=true;
GXInitTexObjCI(obj, GuestToHostPtr(da), (u16)w, (u16)h, (GXCITexFmt)f, (GXTexWrapMode)ws, (GXTexWrapMode)wt, (GXBool)m, tl); MarkHostTexObjConstructed(oa);
// Also write to guest memory so reads work
WriteGuestTexObj(oa, canonicalDataAddr, (u16)w, (u16)h, f, ws, wt, m != 0, true, tl);
// GXInitTexObjCI clears bit1 in the flags byte; keep guest memory consistent.
@@ -306,10 +304,9 @@ extern "C" void GX__InitTexObjCI_80170a04(uint32_t oa, uint32_t da, uint32_t w,
PPC_NATIVE_OVERRIDE_VOID(80170a04, GX__InitTexObjCI_80170a04, (uint32_t oa, uint32_t da, uint32_t w, uint32_t h, uint32_t f, uint32_t ws, uint32_t wt, uint32_t m, uint32_t tl), (oa, da, w, h, f, ws, wt, m, tl));
extern "C" void GX__InitTexObjLOD_80170a4c(uint32_t oa, uint32_t mif, uint32_t maf, float mil, float mal, float lb, uint32_t bc, uint32_t el, uint32_t ma) {
std::lock_guard<std::mutex> guard(g_texObjMutex); GXTexObj* obj = GetHostTexObj(oa); TexObjMeta& meta = GetTexObjMeta(oa);
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
float fmal = std::isfinite(mal) && mal >= 0.f ? mal : 0.f, fmil = std::isfinite(mil) && mil >= 0.f ? std::min(mil, fmal) : 0.f;
meta.minFilter=mif; meta.magFilter=maf; meta.minLod=fmil; meta.maxLod=fmal; meta.lodBias=lb; meta.biasClamp=(bc!=0); meta.edgeLod=(el!=0); meta.maxAniso=ma;
GXInitTexObjLOD(obj, (GXTexFilter)mif, (GXTexFilter)maf, fmil, fmal, lb, (GXBool)bc, (GXBool)el, (GXAnisotropy)ma);
// Also write LOD info to guest memory
WriteGuestTexObjLOD(oa, mif, maf, fmil, fmal, lb, bc != 0, el != 0, ma);
}
@@ -317,7 +314,6 @@ PPC_NATIVE_OVERRIDE_VOID(80170a4c, GX__InitTexObjLOD_80170a4c, (uint32_t oa, uin
extern "C" void GX__InitTexObjWrapMode_80170b50(uint32_t oa, uint32_t ws, uint32_t wt) {
std::lock_guard<std::mutex> guard(g_texObjMutex);
GXTexObj* obj = GetHostTexObj(oa);
TexObjMeta& meta = GetTexObjMeta(oa);
meta.wrapS = ws;
meta.wrapT = wt;
@@ -327,16 +323,13 @@ extern "C" void GX__InitTexObjWrapMode_80170b50(uint32_t oa, uint32_t ws, uint32
Memory::Write32(oa + 0x00, word0);
} catch (...) {
}
GXInitTexObjWrapMode(obj, (GXTexWrapMode)ws, (GXTexWrapMode)wt);
}
PPC_NATIVE_OVERRIDE_VOID(80170b50, GX__InitTexObjWrapMode_80170b50, (uint32_t oa, uint32_t ws, uint32_t wt), (oa, ws, wt));
extern "C" void GX__InitTexObjTlut_80170b64(uint32_t oa, uint32_t tl) {
std::lock_guard<std::mutex> guard(g_texObjMutex);
GXTexObj* obj = GetHostTexObj(oa);
TexObjMeta& meta = GetTexObjMeta(oa);
meta.tlut = tl;
GXInitTexObjTlut(obj, tl);
// Keep guest GXTexObj coherent for later GXLoadTexObj from memory.
try {
Memory::Write32(oa + 0x18, tl);
@@ -365,11 +358,6 @@ extern "C" void GX__InitTexObjFilter_80170b6c(uint32_t oa, uint32_t minFilter, u
Memory::Write32(oa + 0x00, word0);
} catch (...) {}
// Update host texture object if it exists
if (GXTexObj* obj = TryGetHostTexObj(oa)) {
ApplyHostTexObjLod(obj, meta, (GXTexFilter)minFilter, (GXTexFilter)magFilter, meta.lodBias,
(GXAnisotropy)meta.maxAniso);
}
}
PPC_NATIVE_OVERRIDE_VOID(80170b6c, GX__InitTexObjFilter_80170b6c, (uint32_t oa, uint32_t minFilter, uint32_t magFilter), (oa, minFilter, magFilter));
@@ -395,11 +383,6 @@ extern "C" void GX__InitTexObjLODBias_80170b94(uint32_t oa, float bias) {
Memory::Write32(oa + 0x00, word0);
} catch (...) {}
// Update host texture object if it exists
if (GXTexObj* obj = TryGetHostTexObj(oa)) {
ApplyHostTexObjLod(obj, meta, (GXTexFilter)meta.minFilter, (GXTexFilter)meta.magFilter,
clampedBias, (GXAnisotropy)meta.maxAniso);
}
}
PPC_NATIVE_OVERRIDE_VOID(80170b94, GX__InitTexObjLODBias_80170b94, (uint32_t oa, float bias), (oa, bias));
@@ -411,11 +394,6 @@ extern "C" void GX__InitTexObjUserData_80170be8(uint32_t oa, uint32_t userData)
// The RVL SDK stores this opaque guest pointer verbatim at GXTexObj + 0x10.
WriteGuest32(oa + 0x10, userData, "GXInitTexObjUserData");
// Aurora keeps an expanded host-side object, so mirror the pointer when
// that representation has already been constructed.
if (GXTexObj* obj = TryGetHostTexObj(oa)) {
GXInitTexObjUserData(obj, GuestToHostPtr(userData));
}
}
PPC_NATIVE_OVERRIDE_VOID(80170be8, GX__InitTexObjUserData_80170be8,
(uint32_t oa, uint32_t userData), (oa, userData));
@@ -450,7 +428,6 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
static uint32_t s_invalidMetaLogCount = 0;
static uint32_t s_invalidTidLogCount = 0;
static uint32_t s_invalidDimLogCount = 0;
static uint32_t s_hostExceptionLogCount = 0;
static uint32_t s_metaLookupLogCount = 0;
static uint32_t s_unknownFormatLogCount = 0;
static uint32_t s_invalidDataLogCount = 0;
@@ -471,7 +448,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
RT_LOGF(RT_TAG_GX,
"GXLoadTexObj rejected reason=no-metadata oa=0x%08X tid=%u\n", oa, tid);
}
BindUnloadableTexturePlaceholder(tid);
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
return;
}
@@ -481,7 +458,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
"GXLoadTexObj rejected reason=invalid-meta oa=0x%08X tid=%u data=0x%08X %ux%u\n",
oa, tid, meta.dataAddr, meta.width, meta.height);
}
BindUnloadableTexturePlaceholder(tid);
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
return;
}
if (!IsReasonableTextureDimensions(meta.width, meta.height)) {
@@ -490,7 +467,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
"GXLoadTexObj rejected reason=bad-dimensions oa=0x%08X tid=%u %ux%u fmt=0x%X\n",
oa, tid, meta.width, meta.height, meta.format);
}
BindUnloadableTexturePlaceholder(tid);
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
return;
}
// One gate, two reasons: every aurora-loadable format is also a known one,
@@ -506,7 +483,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
known ? "unsupported-format" : "unknown-format", oa, tid, meta.format,
meta.width, meta.height);
}
BindUnloadableTexturePlaceholder(tid);
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
return;
}
const float maxMipLevel = static_cast<float>(ComputeMaxMipLevel(meta.width, meta.height));
@@ -527,7 +504,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
meta.mipmap ? 1u : 0u, static_cast<uint32_t>(maxLod),
cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u);
}
BindUnloadableTexturePlaceholder(tid);
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
return;
}
const GXTexWrapMode wrapS = SanitizeWrapMode(meta.wrapS);
@@ -544,58 +521,140 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
meta.maxLod = maxLodSafe;
meta.lodBias = lodBiasSafe;
// Check if host texture object exists, create if needed
GXTexObj* obj = TryGetHostTexObj(oa);
const bool isPalette = IsPaletteTexFormat(meta.format);
bool needsInit = (obj == nullptr);
const auto metaIt = g_TexObjMeta.find(oa);
if (!needsInit && metaIt != g_TexObjMeta.end()) {
const TexObjMeta& cached = metaIt->second;
if (cached.width != meta.width || cached.height != meta.height || cached.format != meta.format ||
cached.wrapS != meta.wrapS || cached.wrapT != meta.wrapT || cached.mipmap != meta.mipmap ||
cached.minFilter != meta.minFilter || cached.magFilter != meta.magFilter ||
cached.minLod != meta.minLod || cached.maxLod != meta.maxLod ||
cached.lodBias != meta.lodBias || cached.biasClamp != meta.biasClamp ||
cached.edgeLod != meta.edgeLod || cached.maxAniso != meta.maxAniso ||
cached.tlut != meta.tlut || cached.userData != meta.userData) {
needsInit = true;
// The sanitized meta is what the GX thread builds the aurora object from
// and what the next load compares against. The upload flag is consumed
// here so a guest write re-uploads exactly once.
GxTexObjLoad load{};
load.objAddr = oa;
load.tid = tid;
load.upload = meta.needsUpload;
meta.needsUpload = false;
load.meta = meta;
// Write through GetTexObjMeta so the DCStoreRange interval index is
// told this entry's backing may have moved.
GetTexObjMeta(oa) = meta;
GxThread::Post(&GxHostLoadTexObj_gx, load);
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) { Memory::Write32(gd + 0x5FCu, Memory::Read32(gd + 0x5FCu) | 1u); Memory::Write16(gd + 2, 0); } } catch (...) {}
}
PPC_NATIVE_OVERRIDE_VOID(80170f2c, GX__LoadTexObj_80170f2c, (uint32_t oa, uint32_t tid), (oa, tid));
extern "C" void GX__LoadTexObjPreLoaded_80170dc8(uint32_t oa, uint32_t tid) { GX__LoadTexObj_80170f2c(oa, tid); }
PPC_NATIVE_OVERRIDE_VOID(80170dc8, GX__LoadTexObjPreLoaded_80170dc8, (uint32_t oa, uint32_t tid), (oa, tid));
// ============================================================================
// GX-thread side: aurora texture and TLUT objects
// ============================================================================
// Keyed by guest object address like the meta table. Each load carries the
// sanitized meta snapshot; an object is (re)built when it differs from what
// the aurora object was last built from, exactly as the single-threaded
// path compared the cached meta before this split.
namespace aurora::gfx {
struct TextureRef;
using TextureHandle = std::shared_ptr<TextureRef>;
} // namespace aurora::gfx
namespace {
struct HostGXTlutObj {
alignas(GXTlutObj) std::byte publicStorage[sizeof(GXTlutObj)]{};
aurora::gfx::TextureHandle ref;
};
struct HleTlutObj {
static constexpr size_t kTlutObjStorageSize =
(sizeof(GXTlutObj) >= sizeof(HostGXTlutObj)) ? sizeof(GXTlutObj) : sizeof(HostGXTlutObj);
static constexpr size_t kTlutObjStorageAlign =
(alignof(GXTlutObj) >= alignof(HostGXTlutObj)) ? alignof(GXTlutObj) : alignof(HostGXTlutObj);
using Storage = std::aligned_storage_t<kTlutObjStorageSize, kTlutObjStorageAlign>;
Storage storage{};
bool storageLive = false;
bool constructed = false;
HleTlutObj() = default;
~HleTlutObj() { Destroy(); }
HostGXTlutObj* HostObj() { return reinterpret_cast<HostGXTlutObj*>(&storage); }
GXTlutObj* PublicPtr() { return reinterpret_cast<GXTlutObj*>(HostObj()->publicStorage); }
void EnsureStorageLive() {
if (!storageLive) {
new (&storage) HostGXTlutObj();
storageLive = true;
}
}
bool tlutChanged = false;
void Destroy() {
if (storageLive) {
GXDestroyTlutObj(PublicPtr());
std::destroy_at(HostObj());
std::memset(&storage, 0, sizeof(storage));
storageLive = false;
constructed = false;
}
}
};
struct GxHostTexObjEntry {
HleTexObj host;
TexObjMeta cached;
};
struct GxHostTlutObjEntry {
HleTlutObj host;
TlutObjMeta cached;
};
std::unordered_map<uint32_t, GxHostTexObjEntry> g_gxHostTexObjs;
std::map<uint32_t, GxHostTlutObjEntry> g_gxHostTlutObjs;
bool SameTexObjBuildMeta(const TexObjMeta& cached, const TexObjMeta& meta) {
return cached.width == meta.width && cached.height == meta.height && cached.format == meta.format &&
cached.wrapS == meta.wrapS && cached.wrapT == meta.wrapT && cached.mipmap == meta.mipmap &&
cached.minFilter == meta.minFilter && cached.magFilter == meta.magFilter &&
cached.minLod == meta.minLod && cached.maxLod == meta.maxLod &&
cached.lodBias == meta.lodBias && cached.biasClamp == meta.biasClamp &&
cached.edgeLod == meta.edgeLod && cached.maxAniso == meta.maxAniso &&
cached.tlut == meta.tlut && cached.userData == meta.userData;
}
} // namespace
void GxHostBindPlaceholder_gx(uint32_t tid) { BindUnloadableTexturePlaceholder(tid); }
void GxHostLoadTexObj_gx(GxTexObjLoad load) {
static uint32_t s_hostExceptionLogCount = 0;
const TexObjMeta& meta = load.meta;
const uint32_t oa = load.objAddr;
const uint32_t tid = load.tid;
if (tid >= g_boundTexMaps.size()) {
return;
}
const bool isPalette = IsPaletteTexFormat(meta.format);
const uint8_t maxLod = (meta.maxLod > 0.0f) ? ((meta.maxLod > 255.0f) ? 255u : static_cast<uint8_t>(meta.maxLod)) : 0u;
const uint32_t size = GXGetTexBufferSize(meta.width, meta.height, meta.format, (GXBool)meta.mipmap, maxLod);
GxHostTexObjEntry& entry = g_gxHostTexObjs[oa];
GXTexObj* obj = entry.host.constructed ? entry.host.PublicPtr() : nullptr;
const bool needsInit = obj == nullptr || !SameTexObjBuildMeta(entry.cached, meta);
bool textureDataUploaded = false;
try {
if (needsInit) {
if (!obj) {
obj = CreateHostTexObj(oa);
}
entry.host.Destroy();
entry.host.EnsureStorageLive();
obj = entry.host.PublicPtr();
void* dp = GuestToHostPtr(meta.dataAddr, size);
if (dp) {
if (isPalette) {
GXInitTexObjCI(obj, dp, meta.width, meta.height, (GXCITexFmt)meta.format,
wrapS, wrapT,
(GXTexWrapMode)meta.wrapS, (GXTexWrapMode)meta.wrapT,
meta.mipmap ? GX_TRUE : GX_FALSE, meta.tlut);
} else {
GXInitTexObj(obj, dp, meta.width, meta.height, (GXTexFmt)meta.format,
wrapS, wrapT,
(GXTexWrapMode)meta.wrapS, (GXTexWrapMode)meta.wrapT,
meta.mipmap ? GX_TRUE : GX_FALSE);
}
ApplyHostTexObjLod(obj, meta, minFilter, magFilter, meta.lodBias, maxAnisoSafe);
ApplyHostTexObjLod(obj, meta, (GXTexFilter)meta.minFilter, (GXTexFilter)meta.magFilter,
meta.lodBias, (GXAnisotropy)meta.maxAniso);
GXInitTexObjUserData(obj, GuestToHostPtr(meta.userData));
}
MarkHostTexObjConstructed(oa);
// Write through GetTexObjMeta so the DCStoreRange interval index is
// told this entry's backing may have moved.
GetTexObjMeta(oa) = meta;
} else if (isPalette && metaIt != g_TexObjMeta.end() && metaIt->second.tlut != meta.tlut) {
GXInitTexObjTlut(obj, meta.tlut);
GetTexObjMeta(oa).tlut = meta.tlut;
tlutChanged = true;
entry.host.constructed = true;
entry.cached = meta;
}
const void* dp = GuestToHostPtr(meta.dataAddr, size);
if (dp && meta.needsUpload) {
if (dp && load.upload) {
GXInitTexObjData(obj, dp);
meta.needsUpload = false;
textureDataUploaded = true;
}
// Everything the binding contract compares apart from objAddr is the
@@ -607,10 +666,9 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
newBound.dataAddr = CanonicalizeGxMainRamAddress(meta.dataAddr);
const BoundTexInfo& oldBound = g_boundTexMaps[tid];
const bool canSkipHostLoad = GxTextureBindingContract::CanSkipHostLoad(
oldBound, newBound, needsInit, tlutChanged, textureDataUploaded);
oldBound, newBound, needsInit, /*tlutChanged=*/false, textureDataUploaded);
g_boundTexMaps[tid] = newBound;
GetTexObjMeta(oa) = meta;
entry.cached = meta;
if (!canSkipHostLoad) {
GXLoadTexObj(obj, (GXTexMapID)tid);
}
@@ -621,7 +679,6 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
ex.what(), oa, tid, meta.format, meta.width, meta.height, meta.dataAddr);
}
BindUnloadableTexturePlaceholder(tid);
return;
} catch (...) {
if (s_hostExceptionLogCount++ < 64) {
RT_LOGF(RT_TAG_GX,
@@ -629,14 +686,39 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
oa, tid, meta.format, meta.width, meta.height, meta.dataAddr);
}
BindUnloadableTexturePlaceholder(tid);
}
}
void GxHostLoadTlut_gx(GxTlutLoad load) {
GxHostTlutObjEntry& entry = g_gxHostTlutObjs[load.objAddr];
if (!load.valid) {
// GXInitTlutObj refused the descriptor (or the object was never
// initialised): load a fresh empty object rather than read entries*2
// bytes off an unvalidated pointer, as the soft-fail path always did.
static uint32_t s_invalidLogCount = 0;
if (s_invalidLogCount++ < 64) {
RT_LOGF(RT_TAG_GX, "GXTex: invalid GXTlutObj @0x%08X\n", load.objAddr);
}
entry.host.Destroy();
entry.host.EnsureStorageLive();
entry.host.constructed = true;
entry.cached = TlutObjMeta{};
GXLoadTlut(entry.host.PublicPtr(), (GXTlut)load.tlut);
return;
}
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) { Memory::Write32(gd + 0x5FCu, Memory::Read32(gd + 0x5FCu) | 1u); Memory::Write16(gd + 2, 0); } } catch (...) {}
const bool metaChanged = entry.cached.dataAddr != load.meta.dataAddr ||
entry.cached.format != load.meta.format ||
entry.cached.entries != load.meta.entries;
if (!entry.host.constructed || load.rebuild || metaChanged) {
entry.host.Destroy();
entry.host.EnsureStorageLive();
GXInitTlutObj(entry.host.PublicPtr(), GuestToHostPtr(load.meta.dataAddr),
(GXTlutFmt)load.meta.format, load.meta.entries);
entry.host.constructed = true;
entry.cached = load.meta;
}
GXLoadTlut(entry.host.PublicPtr(), (GXTlut)load.tlut);
}
PPC_NATIVE_OVERRIDE_VOID(80170f2c, GX__LoadTexObj_80170f2c, (uint32_t oa, uint32_t tid), (oa, tid));
extern "C" void GX__LoadTexObjPreLoaded_80170dc8(uint32_t oa, uint32_t tid) { GX__LoadTexObj_80170f2c(oa, tid); }
PPC_NATIVE_OVERRIDE_VOID(80170dc8, GX__LoadTexObjPreLoaded_80170dc8, (uint32_t oa, uint32_t tid), (oa, tid));
// ============================================================================
// Texture Object Getters
@@ -676,33 +758,59 @@ PPC_NATIVE_OVERRIDE_VOID(80170cbc, GX__GetTexObjLODAll_80170cbc, (uint32_t oa, u
// ============================================================================
extern "C" void GX__InitTlutObj_80170f80(uint32_t oa, uint32_t da, uint32_t f, uint32_t e) {
std::lock_guard<std::mutex> guard(g_tlutObjMutex); GXTlutObj* obj = CreateHostTlutObj(oa); TlutObjMeta& meta = GetTlutObjMeta(oa);
std::lock_guard<std::mutex> guard(g_tlutObjMutex); TlutObjMeta& meta = GetTlutObjMeta(oa);
meta.dataAddr=CanonicalizeGxMainRamAddress(da); meta.format=f; meta.entries=(u16)e; meta.dirty=false;
// Leave the object unconstructed on a bad descriptor. A later GXLoadTlut
// then takes GetHostTlutObj's soft-fail path and gets a fresh empty object
// instead of aurora reading entries*2 bytes off an unvalidated pointer.
if (!ValidateTlutData(oa, meta)) return;
GXInitTlutObj(obj, GuestToHostPtr(da), (GXTlutFmt)f, (u16)e); MarkHostTlutObjConstructed(oa);
// The aurora object is built by the GX thread on the next GXLoadTlut from
// this meta; a bad descriptor loads an empty object there instead of
// aurora reading entries*2 bytes off an unvalidated pointer.
(void)ValidateTlutData(oa, meta);
}
PPC_NATIVE_OVERRIDE_VOID(80170f80, GX__InitTlutObj_80170f80, (uint32_t oa, uint32_t da, uint32_t f, uint32_t e), (oa, da, f, e));
extern "C" void GX__LoadTlut_80170fa8(uint32_t oa, uint32_t tl) { std::lock_guard<std::mutex> guard(g_tlutObjMutex); if (tl >= kMaxTluts) { RT_LOGF(RT_TAG_GX, "GXLoadTlut: invalid TLUT index %u (oa=0x%08X)\n", tl, oa); return; } auto metaIt = g_TlutObjMeta.find(oa); if (metaIt != g_TlutObjMeta.end() && metaIt->second.dirty) { if (ValidateTlutData(oa, metaIt->second)) { GXTlutObj* rebuild = CreateHostTlutObj(oa); GXInitTlutObj(rebuild, GuestToHostPtr(metaIt->second.dataAddr), (GXTlutFmt)metaIt->second.format, metaIt->second.entries); MarkHostTlutObjConstructed(oa); } metaIt->second.dirty = false; } GXTlutObj* obj = GetHostTlutObj(oa); GXLoadTlut(obj, (GXTlut)tl); try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {} }
extern "C" void GX__LoadTlut_80170fa8(uint32_t oa, uint32_t tl) {
std::lock_guard<std::mutex> guard(g_tlutObjMutex);
if (tl >= kMaxTluts) {
RT_LOGF(RT_TAG_GX, "GXLoadTlut: invalid TLUT index %u (oa=0x%08X)\n", tl, oa);
return;
}
GxTlutLoad load{};
load.objAddr = oa;
load.tlut = tl;
auto metaIt = g_TlutObjMeta.find(oa);
if (metaIt != g_TlutObjMeta.end()) {
TlutObjMeta& meta = metaIt->second;
load.valid = ValidateTlutData(oa, meta);
// A guest write over the palette (DCStoreRange) marks it dirty; the
// rebuild travels with this load and the flag is consumed once.
load.rebuild = meta.dirty && load.valid;
meta.dirty = false;
load.meta = meta;
}
GxThread::Post(&GxHostLoadTlut_gx, load);
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {}
}
PPC_NATIVE_OVERRIDE_VOID(80170fa8, GX__LoadTlut_80170fa8, (uint32_t oa, uint32_t tl), (oa, tl));
// ============================================================================
// Texture Invalidation and Coordinate Control
// ============================================================================
extern "C" void GX__InvalidateTexAll_80171110() {
static void GX__InvalidateTexAll_80171110_gx() {
// Real GX invalidates its internal texture cache here. Aurora forwards the
// invalidate to the renderer, which keeps unchanged source uploads hot but
// revalidates reused guest buffers before serving cached texture handles.
GXInvalidateTexAll();
}
PPC_NATIVE_OVERRIDE_VOID(80171110, GX__InvalidateTexAll_80171110, (), ());
GX_DEFERRED_OVERRIDE_VOID(80171110, GX__InvalidateTexAll_80171110, (), ());
extern "C" void GX__SetTexCoordScaleManually_80171180(uint32_t c, uint32_t en, uint32_t ss, uint32_t ts) { GXSetTexCoordScaleManually((GXTexCoordID)c, (GXBool)en, (u16)ss, (u16)ts); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ Memory::Write32(gd+0x5E4u, (Memory::Read32(gd+0x5E4u)&~(1u<<c))|((en&1u)<<c)); if(en){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFF0000u)|((ss-1)&0xFFFFu)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFF0000u)|((ts-1)&0xFFFFu)); Memory::Write16(gd+2, 0); } } }catch(...){} }
static void GX__SetTexCoordScaleManually_gx(uint32_t c, uint32_t en, uint32_t ss, uint32_t ts) {
GXSetTexCoordScaleManually((GXTexCoordID)c, (GXBool)en, (u16)ss, (u16)ts);
}
extern "C" void GX__SetTexCoordScaleManually_80171180(uint32_t c, uint32_t en, uint32_t ss, uint32_t ts) { GxThread::Post(&GX__SetTexCoordScaleManually_gx, c, en, ss, ts); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ Memory::Write32(gd+0x5E4u, (Memory::Read32(gd+0x5E4u)&~(1u<<c))|((en&1u)<<c)); if(en){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFF0000u)|((ss-1)&0xFFFFu)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFF0000u)|((ts-1)&0xFFFFu)); Memory::Write16(gd+2, 0); } } }catch(...){} }
PPC_NATIVE_OVERRIDE_VOID(80171180, GX__SetTexCoordScaleManually_80171180, (uint32_t c, uint32_t en, uint32_t ss, uint32_t ts), (c, en, ss, ts));
extern "C" void GX__SetTexCoordBias_801711fc(uint32_t c, uint32_t se, uint32_t te) { GXSetTexCoordBias((GXTexCoordID)c, (GXBool)se, (GXBool)te); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFEFFFFu)|((se&1u)<<16)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFEFFFFu)|((te&1u)<<16)); if(Memory::Read32(gd+0x5E4u)&(1u<<c)) Memory::Write16(gd+2, 0); } }catch(...){} }
static void GX__SetTexCoordBias_gx(uint32_t c, uint32_t se, uint32_t te) {
GXSetTexCoordBias((GXTexCoordID)c, (GXBool)se, (GXBool)te);
}
extern "C" void GX__SetTexCoordBias_801711fc(uint32_t c, uint32_t se, uint32_t te) { GxThread::Post(&GX__SetTexCoordBias_gx, c, se, te); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFEFFFFu)|((se&1u)<<16)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFEFFFFu)|((te&1u)<<16)); if(Memory::Read32(gd+0x5E4u)&(1u<<c)) Memory::Write16(gd+2, 0); } }catch(...){} }
PPC_NATIVE_OVERRIDE_VOID(801711fc, GX__SetTexCoordBias_801711fc, (uint32_t c, uint32_t se, uint32_t te), (c, se, te));
+369
View File
@@ -0,0 +1,369 @@
#include "gx_thread.h"
#include "runtime_log.h"
#include <atomic>
#include <chrono>
#include <condition_variable>
#include <cstdio>
#include <cstdlib>
#include <exception>
#include <mutex>
#include <thread>
#if defined(_WIN32)
#include <windows.h>
#else
#include <pthread.h>
#include <unistd.h>
#if defined(__linux__)
#include <sys/syscall.h>
#endif
#endif
// The ring: one producer (the game thread) and one consumer. Records are
// 16-byte headers followed by a payload, 16-byte aligned, always contiguous;
// a header with a null invoke is a wrap marker that skips to the ring start.
extern "C" void GxFifoConsumeBytes(const uint8_t* data, uint32_t sizeBytes);
namespace GxThread {
namespace {
constexpr uint32_t kRingBytes = 16u << 20;
constexpr uint32_t kRingMask = kRingBytes - 1u;
constexpr uint32_t kHeaderBytes = 16;
constexpr uint32_t kFifoChunkBytes = 8192;
constexpr uint32_t kConsumerSpinIterations = 4000;
struct Header {
uint64_t invoke;
uint32_t payloadBytes;
uint32_t stride;
};
using Clock = std::chrono::steady_clock;
bool g_requested = false;
bool g_enabled = false;
bool g_started = false;
std::thread g_thread;
uint8_t* g_ring = nullptr;
thread_local bool t_isGxThread = false;
alignas(64) std::atomic<uint64_t> g_head{0};
alignas(64) std::atomic<uint64_t> g_tail{0};
alignas(64) std::atomic<bool> g_stop{false};
std::atomic<bool> g_consumerSleeping{false};
std::atomic<bool> g_producerWaiting{false};
std::atomic<uint64_t> g_fenceCompleted{0};
std::atomic<uint32_t> g_nativeTid{0};
std::mutex g_mutex;
std::condition_variable g_cvData;
std::condition_variable g_cvSpace;
std::condition_variable g_cvFence;
void (*g_waitCallback)() = nullptr;
// Producer-only state.
uint64_t g_localTail = 0;
uint64_t g_fenceRequested = 0;
uint8_t g_pendingFifo[kFifoChunkBytes];
uint32_t g_pendingFifoBytes = 0;
// Window statistics. Producer-side counters are plain; consumer-side ones are
// relaxed atomics read by the producer when it formats the log line.
uint64_t g_statRecords = 0;
uint64_t g_statFifoRecords = 0;
uint64_t g_statFifoBytes = 0;
uint64_t g_statBytes = 0;
uint64_t g_statDrains = 0;
uint64_t g_statDrainWaitNs = 0;
uint64_t g_statSpaceWaitNs = 0;
uint64_t g_statQueuePeakBytes = 0;
std::atomic<uint64_t> g_statBusyNs{0};
std::atomic<uint64_t> g_statFaults{0};
uint64_t ElapsedNs(Clock::time_point since) {
return static_cast<uint64_t>(std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - since).count());
}
uint32_t Align16(uint32_t bytes) { return (bytes + 15u) & ~15u; }
void SetThreadName() {
#if defined(_WIN32)
using SetThreadDescriptionFn = HRESULT(WINAPI*)(HANDLE, PCWSTR);
if (HMODULE kernel = ::GetModuleHandleW(L"kernel32.dll")) {
if (auto fn = reinterpret_cast<SetThreadDescriptionFn>(::GetProcAddress(kernel, "SetThreadDescription"))) {
fn(::GetCurrentThread(), L"MKW GX");
}
}
g_nativeTid.store(static_cast<uint32_t>(::GetCurrentThreadId()), std::memory_order_release);
#else
#if defined(__APPLE__)
pthread_setname_np("MKW GX");
#else
pthread_setname_np(pthread_self(), "MKW GX");
#endif
#if defined(__linux__)
g_nativeTid.store(static_cast<uint32_t>(syscall(SYS_gettid)), std::memory_order_release);
#else
g_nativeTid.store(1u, std::memory_order_release);
#endif
#endif
}
// Waits on `cv` until pred() holds, running the wait callback between
// bounded waits so the game thread's VI/alarm servicing never starves.
template <typename Pred>
void WaitWithCallback(std::condition_variable& cv, std::atomic<bool>& waitingFlag, Pred pred) {
waitingFlag.store(true, std::memory_order_release);
while (true) {
{
std::unique_lock<std::mutex> lock(g_mutex);
if (cv.wait_for(lock, std::chrono::milliseconds(2), pred)) {
break;
}
}
if (g_waitCallback != nullptr) {
g_waitCallback();
}
}
waitingFlag.store(false, std::memory_order_release);
}
void NotifyConsumer() {
if (g_consumerSleeping.load(std::memory_order_acquire)) {
std::lock_guard<std::mutex> lock(g_mutex);
g_cvData.notify_one();
}
}
void PostRecordRaw(detail::Invoke invoke, const void* payload, uint32_t payloadBytes) {
const uint32_t stride = Align16(kHeaderBytes + payloadBytes);
uint32_t offset = static_cast<uint32_t>(g_localTail & kRingMask);
const uint32_t wrapBytes = (offset + stride > kRingBytes) ? (kRingBytes - offset) : 0u;
const uint64_t required = static_cast<uint64_t>(stride) + wrapBytes;
const auto freeBytes = [&] { return kRingBytes - (g_localTail - g_head.load(std::memory_order_acquire)); };
if (freeBytes() < required) {
const auto started = Clock::now();
WaitWithCallback(g_cvSpace, g_producerWaiting, [&] { return freeBytes() >= required; });
g_statSpaceWaitNs += ElapsedNs(started);
}
if (wrapBytes != 0) {
const Header wrap{0, 0, wrapBytes};
std::memcpy(g_ring + offset, &wrap, sizeof(wrap));
g_localTail += wrapBytes;
offset = 0;
}
const Header header{reinterpret_cast<uint64_t>(invoke), payloadBytes, stride};
std::memcpy(g_ring + offset, &header, sizeof(header));
if (payloadBytes != 0) {
std::memcpy(g_ring + offset + kHeaderBytes, payload, payloadBytes);
}
g_localTail += stride;
g_tail.store(g_localTail, std::memory_order_release);
++g_statRecords;
g_statBytes += stride;
const uint64_t queued = g_localTail - g_head.load(std::memory_order_relaxed);
if (queued > g_statQueuePeakBytes) {
g_statQueuePeakBytes = queued;
}
NotifyConsumer();
}
void FifoInvoke(const uint8_t* payload, uint32_t payloadBytes) { GxFifoConsumeBytes(payload, payloadBytes); }
void FenceInvoke(const uint8_t* payload, uint32_t) {
uint64_t sequence = 0;
std::memcpy(&sequence, payload, sizeof(sequence));
{
std::lock_guard<std::mutex> lock(g_mutex);
g_fenceCompleted.store(sequence, std::memory_order_release);
}
g_cvFence.notify_all();
}
void ConsumerLoop() {
t_isGxThread = true;
SetThreadName();
uint64_t head = g_head.load(std::memory_order_relaxed);
uint32_t spins = 0;
while (true) {
const uint64_t tail = g_tail.load(std::memory_order_acquire);
if (head == tail) {
if (g_stop.load(std::memory_order_acquire)) {
break;
}
if (spins < kConsumerSpinIterations) {
++spins;
std::this_thread::yield();
continue;
}
g_consumerSleeping.store(true, std::memory_order_release);
{
std::unique_lock<std::mutex> lock(g_mutex);
g_cvData.wait_for(lock, std::chrono::milliseconds(1), [&] {
return g_tail.load(std::memory_order_acquire) != head || g_stop.load(std::memory_order_acquire);
});
}
g_consumerSleeping.store(false, std::memory_order_release);
continue;
}
spins = 0;
Header header;
std::memcpy(&header, g_ring + (head & kRingMask), sizeof(header));
if (header.invoke != 0) {
const auto started = Clock::now();
const uint8_t* payload = g_ring + ((head + kHeaderBytes) & kRingMask);
try {
reinterpret_cast<detail::Invoke>(header.invoke)(payload, header.payloadBytes);
} catch (const std::exception& ex) {
const uint64_t faults = g_statFaults.fetch_add(1u, std::memory_order_relaxed) + 1u;
if (faults <= 64) {
RT_LOGF(RT_TAG_GX, "GX thread: command raised '%s' (n=%llu)\n", ex.what(),
static_cast<unsigned long long>(faults));
}
} catch (...) {
const uint64_t faults = g_statFaults.fetch_add(1u, std::memory_order_relaxed) + 1u;
if (faults <= 64) {
RT_LOGF(RT_TAG_GX, "GX thread: command raised an unknown exception (n=%llu)\n",
static_cast<unsigned long long>(faults));
}
}
g_statBusyNs.fetch_add(ElapsedNs(started), std::memory_order_relaxed);
}
head += header.stride;
g_head.store(head, std::memory_order_release);
if (g_producerWaiting.load(std::memory_order_acquire)) {
std::lock_guard<std::mutex> lock(g_mutex);
g_cvSpace.notify_one();
}
}
t_isGxThread = false;
}
} // namespace
void Configure(bool enabled) { g_requested = enabled; }
bool Enabled() noexcept { return g_enabled; }
bool IsGxThread() noexcept { return t_isGxThread; }
void SetWaitCallback(void (*callback)()) { g_waitCallback = callback; }
uint32_t NativeThreadId() noexcept { return g_nativeTid.load(std::memory_order_acquire); }
void Start() {
if (!g_requested || g_started) {
return;
}
g_ring = static_cast<uint8_t*>(std::malloc(kRingBytes));
if (g_ring == nullptr) {
RT_LOGF(RT_TAG_GX, "GX thread: ring allocation failed; running the GX pipeline on the game thread\n");
return;
}
g_started = true;
g_enabled = true;
g_stop.store(false, std::memory_order_release);
g_thread = std::thread(ConsumerLoop);
RT_LOGF(RT_TAG_GX, "GX thread started (%u MiB command ring)\n", kRingBytes >> 20);
}
void Stop() {
if (!g_started) {
return;
}
Drain();
{
std::lock_guard<std::mutex> lock(g_mutex);
g_stop.store(true, std::memory_order_release);
}
g_cvData.notify_all();
g_thread.join();
g_started = false;
g_enabled = false;
}
void FlushFifo() {
if (g_pendingFifoBytes == 0) {
return;
}
const uint32_t bytes = g_pendingFifoBytes;
g_pendingFifoBytes = 0;
++g_statFifoRecords;
g_statFifoBytes += bytes;
PostRecordRaw(&FifoInvoke, g_pendingFifo, bytes);
}
void Drain() {
if (!g_enabled || !g_started || t_isGxThread) {
return;
}
FlushFifo();
const uint64_t sequence = ++g_fenceRequested;
PostRecordRaw(&FenceInvoke, &sequence, sizeof(sequence));
++g_statDrains;
if (g_fenceCompleted.load(std::memory_order_acquire) >= sequence) {
return;
}
const auto started = Clock::now();
WaitWithCallback(g_cvFence, g_producerWaiting,
[&] { return g_fenceCompleted.load(std::memory_order_acquire) >= sequence; });
g_statDrainWaitNs += ElapsedNs(started);
}
void PostFifoWord(uint32_t value, uint32_t sizeBytes) {
if (sizeBytes == 0 || sizeBytes > 4) {
sizeBytes = 4;
}
if (g_pendingFifoBytes + sizeBytes > kFifoChunkBytes) {
FlushFifo();
}
uint8_t* out = g_pendingFifo + g_pendingFifoBytes;
for (uint32_t i = 0; i < sizeBytes; ++i) {
out[i] = static_cast<uint8_t>(value >> (8u * (sizeBytes - 1u - i)));
}
g_pendingFifoBytes += sizeBytes;
}
void PostFifoBytes(const uint8_t* data, uint32_t sizeBytes) {
if (data == nullptr || sizeBytes == 0) {
return;
}
if (g_pendingFifoBytes + sizeBytes > kFifoChunkBytes) {
FlushFifo();
}
if (sizeBytes > kFifoChunkBytes) {
++g_statFifoRecords;
g_statFifoBytes += sizeBytes;
PostRecordRaw(&FifoInvoke, data, sizeBytes);
return;
}
std::memcpy(g_pendingFifo + g_pendingFifoBytes, data, sizeBytes);
g_pendingFifoBytes += sizeBytes;
}
std::string FormatStatsAndReset(double windowSeconds, uint32_t frames) {
const double perFrame = frames != 0 ? 1.0 / static_cast<double>(frames) : 0.0;
const double busyNs = static_cast<double>(g_statBusyNs.exchange(0, std::memory_order_relaxed));
const double busyPercent = windowSeconds > 0.0 ? busyNs / (windowSeconds * 1e9) * 100.0 : 0.0;
char buffer[512];
std::snprintf(buffer, sizeof(buffer),
"GX thread: %.0f records/frame (%.1f KiB, %.0f FIFO chunks with %.1f KiB), queue peak %.1f KiB; "
"game thread waited %.2f ms/frame for ring space and %.2f ms/frame in %.1f drains/frame; "
"GX thread busy %.1f%%; faults %llu",
static_cast<double>(g_statRecords) * perFrame,
static_cast<double>(g_statBytes) * perFrame / 1024.0,
static_cast<double>(g_statFifoRecords) * perFrame,
static_cast<double>(g_statFifoBytes) * perFrame / 1024.0,
static_cast<double>(g_statQueuePeakBytes) / 1024.0,
static_cast<double>(g_statSpaceWaitNs) * perFrame / 1e6,
static_cast<double>(g_statDrainWaitNs) * perFrame / 1e6,
static_cast<double>(g_statDrains) * perFrame, busyPercent,
static_cast<unsigned long long>(g_statFaults.load(std::memory_order_relaxed)));
g_statRecords = g_statFifoRecords = g_statFifoBytes = g_statBytes = 0;
g_statDrains = g_statDrainWaitNs = g_statSpaceWaitNs = g_statQueuePeakBytes = 0;
return buffer;
}
namespace detail {
void PostRecord(Invoke invoke, const void* payload, uint32_t payloadBytes) {
FlushFifo();
PostRecordRaw(invoke, payload, payloadBytes);
}
} // namespace detail
} // namespace GxThread
+50 -24
View File
@@ -29,6 +29,30 @@ namespace {
std::memcpy(g_projectionVector, projV, sizeof(g_projectionVector));
}
// The SDK copies matrices into the FIFO at call time, so immediate loads
// are snapshotted on the game thread; indexed loads read guest memory when
// the GX thread reaches them, as the GP does.
struct GxMtxSnapshot {
float m[16];
};
static void GX__SetViewportJitter_gx(float l, float t, float w, float h, float nz, float fz, uint32_t f) {
GXSetViewportJitter(l, t, w, h, nz, fz, f);
}
static void GX__SetViewport_gx(float l, float t, float w, float h, float nz, float fz) {
GXSetViewport(l, t, w, h, nz, fz);
}
static void GX__SetZScaleOffset_gx(float s, float o) { GXSetZScaleOffset(s, o); }
static void GX__SetScissorBoxOffset_gx(int32_t xo, int32_t yo) { GXSetScissorBoxOffset(xo, yo); }
static void GX__SetScissor_gx(uint32_t l, uint32_t t, uint32_t w, uint32_t h) { GXSetScissor(l, t, w, h); }
static void GX__SetProjection_gx(GxMtxSnapshot proj, uint32_t pt) {
GXSetProjection(proj.m, (GXProjectionType)pt);
}
static void GX__LoadPosMtxImm_gx(GxMtxSnapshot mtx, uint32_t id) { GXLoadPosMtxImm((float(*)[4])mtx.m, id); }
static void GX__LoadNrmMtxImm_gx(GxMtxSnapshot mtx, uint32_t id) { GXLoadNrmMtxImm((float(*)[4])mtx.m, id); }
static void GX__LoadTexMtxImm_gx(GxMtxSnapshot mtx, uint32_t id, uint32_t t) {
GXLoadTexMtxImm(mtx.m, id, (GXTexMtxType)t);
}
}
// ============================================================================
@@ -40,13 +64,13 @@ namespace {
// transform double-transformed menu viewports and broke 640x480 overscan, so it was removed.
extern "C" void GX__SetViewportJitter_80173378(float l, float t, float w, float h, float nz, float fz, uint32_t f) {
g_viewportState[0]=l; g_viewportState[1]=t; g_viewportState[2]=w; g_viewportState[3]=h; g_viewportState[4]=nz; g_viewportState[5]=fz;
GXSetViewportJitter(l, t, w, h, nz, fz, f);
GxThread::Post(&GX__SetViewportJitter_gx, l, t, w, h, nz, fz, f);
}
PPC_NATIVE_OVERRIDE_VOID(80173378, GX__SetViewportJitter_80173378, (float l, float t, float w, float h, float nz, float fz, uint32_t f), (l, t, w, h, nz, fz, f));
extern "C" void GX__SetViewport_801733b4(float l, float t, float w, float h, float nz, float fz) {
g_viewportState[0]=l; g_viewportState[1]=t; g_viewportState[2]=w; g_viewportState[3]=h; g_viewportState[4]=nz; g_viewportState[5]=fz;
GXSetViewport(l, t, w, h, nz, fz);
GxThread::Post(&GX__SetViewport_gx, l, t, w, h, nz, fz);
}
PPC_NATIVE_OVERRIDE_VOID(801733b4, GX__SetViewport_801733b4, (float l, float t, float w, float h, float nz, float fz), (l, t, w, h, nz, fz));
@@ -66,7 +90,7 @@ extern "C" void GX__GetViewportv_801733e0(uint32_t oa) {
PPC_NATIVE_OVERRIDE_VOID(801733e0, GX__GetViewportv_801733e0, (uint32_t oa), (oa));
extern "C" void GX__SetZScaleOffset_80173400(float s, float o) {
GXSetZScaleOffset(s, o);
GxThread::Post(&GX__SetZScaleOffset_gx, s, o);
try {
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
if (gd) {
@@ -80,7 +104,7 @@ extern "C" void GX__SetZScaleOffset_80173400(float s, float o) {
PPC_NATIVE_OVERRIDE_VOID(80173400, GX__SetZScaleOffset_80173400, (float s, float o), (s, o));
extern "C" void GX__SetScissorBoxOffset_801734e0(int32_t xo, int32_t yo) {
GXSetScissorBoxOffset(xo, yo);
GxThread::Post(&GX__SetScissorBoxOffset_gx, xo, yo);
try {
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
if (gd) Memory::Write16(gd + 2, 0);
@@ -104,7 +128,7 @@ extern "C" void GX__SetScissor_80173430(uint32_t l, uint32_t t, uint32_t w, uint
// No viewport replay here: aurora recomputes viewport and scissor together
// on every scissor change (set_logical_scissor -> apply_logical_render_state),
// so re-issuing the current viewport would be duplicate work.
GXSetScissor(l, t, w, h);
GxThread::Post(&GX__SetScissor_gx, l, t, w, h);
}
PPC_NATIVE_OVERRIDE_VOID(80173430, GX__SetScissor_80173430, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
@@ -113,10 +137,10 @@ PPC_NATIVE_OVERRIDE_VOID(80173430, GX__SetScissor_80173430, (uint32_t l, uint32_
// ============================================================================
extern "C" void GX__SetProjection_8017301c(uint32_t ma, uint32_t pt) {
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 64); float m[16];
SwapBeF32ArrayToHost(raw, m, 16);
GXSetProjection(m, (GXProjectionType)pt);
UpdateProjectionVectorFromMatrix(m, (GXProjectionType)pt);
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 64); GxMtxSnapshot proj{};
SwapBeF32ArrayToHost(raw, proj.m, 16);
UpdateProjectionVectorFromMatrix(proj.m, (GXProjectionType)pt);
GxThread::Post(&GX__SetProjection_gx, proj, pt);
}
PPC_NATIVE_OVERRIDE_VOID(8017301c, GX__SetProjection_8017301c, (uint32_t ma, uint32_t pt), (ma, pt));
@@ -124,10 +148,10 @@ extern "C" void GX__SetProjectionv_80173080(uint32_t pa) {
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(pa, 28); float v[7];
SwapBeF32ArrayToHost(raw, v, 7);
GXProjectionType pt=(v[0]!=0.f)?GX_ORTHOGRAPHIC:GX_PERSPECTIVE;
float m[16]={0.f}; m[0]=v[1]; m[5]=v[3]; m[10]=v[5]; m[11]=v[6];
GxMtxSnapshot proj{}; float* m = proj.m; m[0]=v[1]; m[5]=v[3]; m[10]=v[5]; m[11]=v[6];
if(pt==GX_PERSPECTIVE){ m[2]=v[2]; m[6]=v[4]; m[14]=-1.f; } else { m[3]=v[2]; m[7]=v[4]; m[15]=1.f; }
GXSetProjection(m, pt);
UpdateProjectionVectorFromProjV(v);
GxThread::Post(&GX__SetProjection_gx, proj, static_cast<uint32_t>(pt));
}
PPC_NATIVE_OVERRIDE_VOID(80173080, GX__SetProjectionv_80173080, (uint32_t pa), (pa));
@@ -144,27 +168,28 @@ PPC_NATIVE_OVERRIDE_VOID(801730cc, GX__GetProjectionv_801730cc, (uint32_t pa), (
// ============================================================================
extern "C" void GX__LoadPosMtxImm_8017310c(uint32_t ma, uint32_t id) {
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma); float m[12];
SwapBeF32ArrayToHost(raw, m, 12);
GXLoadPosMtxImm((float(*)[4])m, id);
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma); GxMtxSnapshot mtx{};
SwapBeF32ArrayToHost(raw, mtx.m, 12);
GxThread::Post(&GX__LoadPosMtxImm_gx, mtx, id);
}
PPC_NATIVE_OVERRIDE_VOID(8017310c, GX__LoadPosMtxImm_8017310c, (uint32_t ma, uint32_t id), (ma, id));
extern "C" void GX__LoadPosMtxIndx_8017315c(uint32_t mi, uint32_t id) {
static void GX__LoadPosMtxIndx_8017315c_gx(uint32_t mi, uint32_t id) {
const auto& arr=g_hleGxState.vtxArray[GX_POS_MTX_ARRAY];
if(arr.base==0||arr.stride==0) return;
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(arr.base+mi*arr.stride, 48);
if(raw){ float m[12]; SwapBeF32ArrayToHost(raw,m,12); GXLoadPosMtxImm((float(*)[4])m,id); }
}
PPC_NATIVE_OVERRIDE_VOID(8017315c, GX__LoadPosMtxIndx_8017315c, (uint32_t mi, uint32_t id), (mi, id));
GX_DEFERRED_OVERRIDE_VOID(8017315c, GX__LoadPosMtxIndx_8017315c, (uint32_t mi, uint32_t id), (mi, id));
extern "C" void GX__LoadNrmMtxImm_80173188(uint32_t ma, uint32_t id) {
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 48); float m[12];
SwapBeF32ArrayToHost(raw, m, 12); GXLoadNrmMtxImm((float(*)[4])m, id);
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 48); GxMtxSnapshot mtx{};
SwapBeF32ArrayToHost(raw, mtx.m, 12);
GxThread::Post(&GX__LoadNrmMtxImm_gx, mtx, id);
}
PPC_NATIVE_OVERRIDE_VOID(80173188, GX__LoadNrmMtxImm_80173188, (uint32_t ma, uint32_t id), (ma, id));
extern "C" void GX__LoadNrmMtxIndx3x3_801731e0(uint32_t mi, uint32_t id) {
static void GX__LoadNrmMtxIndx3x3_801731e0_gx(uint32_t mi, uint32_t id) {
const auto& arr=g_hleGxState.vtxArray[GX_NRM_MTX_ARRAY];
if(arr.base==0||arr.stride==0) return;
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(arr.base+mi*arr.stride, 36);
@@ -180,14 +205,15 @@ extern "C" void GX__LoadNrmMtxIndx3x3_801731e0(uint32_t mi, uint32_t id) {
GXLoadNrmMtxImm((float(*)[4])m, id);
}
}
PPC_NATIVE_OVERRIDE_VOID(801731e0, GX__LoadNrmMtxIndx3x3_801731e0, (uint32_t mi, uint32_t id), (mi, id));
GX_DEFERRED_OVERRIDE_VOID(801731e0, GX__LoadNrmMtxIndx3x3_801731e0, (uint32_t mi, uint32_t id), (mi, id));
extern "C" void GX__LoadTexMtxImm_80173234(uint32_t ma, uint32_t id, uint32_t t) {
size_t c=(t==(uint32_t)GX_MTX3x4)?12:8;
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma,c*4); float l[12]={};
SwapBeF32ArrayToHost(raw,l,c); GXLoadTexMtxImm(l, id, (GXTexMtxType)t);
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma,c*4); GxMtxSnapshot mtx{};
SwapBeF32ArrayToHost(raw, mtx.m, c);
GxThread::Post(&GX__LoadTexMtxImm_gx, mtx, id, t);
}
PPC_NATIVE_OVERRIDE_VOID(80173234, GX__LoadTexMtxImm_80173234, (uint32_t ma, uint32_t id, uint32_t t), (ma, id, t));
extern "C" void GX__SetCurrentMtx_80173214(uint32_t id) { GXSetCurrentMtx(id); }
PPC_NATIVE_OVERRIDE_VOID(80173214, GX__SetCurrentMtx_80173214, (uint32_t id), (id));
static void GX__SetCurrentMtx_80173214_gx(uint32_t id) { GXSetCurrentMtx(id); }
GX_DEFERRED_OVERRIDE_VOID(80173214, GX__SetCurrentMtx_80173214, (uint32_t id), (id));
+75 -37
View File
@@ -8,6 +8,10 @@
#include <cstdio>
#include <cstdlib>
// Game-thread mirror for the GXGet* overrides; the parser state in
// g_hleGxState belongs to the GX thread.
GxGameSideVertexState g_gxGameVertexState;
namespace {
uint32_t CanonicalVtxAttr(uint32_t attr) {
return attr == GX_VA_NBT ? GX_VA_NRM : attr;
@@ -31,7 +35,7 @@ void ApplyAuroraVtxStateForBegin(GXVtxFmt fmt) {
// Vertex Descriptor
// ============================================================================
extern "C" void GX__ClearVtxDesc_8016dc34() {
static void GX__ClearVtxDesc_8016dc34_gx() {
bool changed = false;
for(int i=0; i<26; ++i){
changed |= g_hleGxState.vtxDesc[i] != GX_NONE;
@@ -41,9 +45,15 @@ extern "C" void GX__ClearVtxDesc_8016dc34() {
// GXClearVtxDesc resets descriptors only; array base/stride state persists.
GXClearVtxDesc();
}
extern "C" void GX__ClearVtxDesc_8016dc34() {
for (int i = 0; i < 26; ++i) {
g_gxGameVertexState.vtxDesc[i] = GX_NONE;
}
GxThread::Post(&GX__ClearVtxDesc_8016dc34_gx);
}
PPC_NATIVE_OVERRIDE_VOID(8016dc34, GX__ClearVtxDesc_8016dc34, (), ());
extern "C" void GX__SetVtxDesc_8016d3a4(uint32_t a, uint32_t t) {
static void GX__SetVtxDesc_8016d3a4_gx(uint32_t a, uint32_t t) {
const uint32_t attr = CanonicalVtxAttr(a);
if(attr>=26||attr==GX_VA_NULL) return;
const GXAttrType oldType = g_hleGxState.vtxDesc[attr];
@@ -55,6 +65,16 @@ extern "C" void GX__SetVtxDesc_8016d3a4(uint32_t a, uint32_t t) {
if(IsMatrixIndexAttr((GXAttr)attr)) return;
GXSetVtxDesc((GXAttr)a, (t==GX_INDEX8||t==GX_INDEX16)?GX_DIRECT:(GXAttrType)t);
}
extern "C" void GX__SetVtxDesc_8016d3a4(uint32_t a, uint32_t t) {
const uint32_t attr = CanonicalVtxAttr(a);
if (attr < 26 && attr != GX_VA_NULL) {
g_gxGameVertexState.vtxDesc[attr] = (GXAttrType)t;
if (a == GX_VA_NBT) {
g_gxGameVertexState.vtxDesc[GX_VA_NBT] = GX_NONE;
}
}
GxThread::Post(&GX__SetVtxDesc_8016d3a4_gx, a, t);
}
PPC_NATIVE_OVERRIDE_VOID(8016d3a4, GX__SetVtxDesc_8016d3a4, (uint32_t a, uint32_t t), (a, t));
extern "C" void GX__SetVtxDescv_8016d608(uint32_t la) {
@@ -72,7 +92,7 @@ extern "C" void GX__GetVtxDesc_8016d9f0(uint32_t a, uint32_t tp) {
GXAttrType type = GX_NONE;
const uint32_t attr = CanonicalVtxAttr(a);
if (attr < GX_VA_MAX_ATTR && attr != GX_VA_NULL) {
type = g_hleGxState.vtxDesc[attr];
type = g_gxGameVertexState.vtxDesc[attr];
}
if (tp) {
Memory::Write32(tp, static_cast<uint32_t>(type));
@@ -88,11 +108,11 @@ extern "C" void GX__GetVtxDescv_8016dba4(uint32_t la) {
uint32_t p = la;
for (uint32_t a = GX_VA_PNMTXIDX; a <= GX_VA_TEX7; ++a) {
Memory::Write32(p, a);
Memory::Write32(p + 4, static_cast<uint32_t>(g_hleGxState.vtxDesc[a]));
Memory::Write32(p + 4, static_cast<uint32_t>(g_gxGameVertexState.vtxDesc[a]));
p += 8;
}
Memory::Write32(p, GX_VA_NBT);
Memory::Write32(p + 4, static_cast<uint32_t>(g_hleGxState.vtxDesc[GX_VA_NRM]));
Memory::Write32(p + 4, static_cast<uint32_t>(g_gxGameVertexState.vtxDesc[GX_VA_NRM]));
p += 8;
Memory::Write32(p, GX_VA_NULL);
}
@@ -102,7 +122,7 @@ PPC_NATIVE_OVERRIDE_VOID(8016dba4, GX__GetVtxDescv_8016dba4, (uint32_t la), (la)
// Vertex Attribute Format
// ============================================================================
extern "C" void GX__SetVtxAttrFmt_8016dc68(uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr) {
static void GX__SetVtxAttrFmt_8016dc68_gx(uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr) {
const uint32_t attr = CanonicalVtxAttr(a);
if(vf<8&&attr<26){
const VtxAttrFmt oldFmt = g_hleGxState.vtxAttrFmt[vf][attr];
@@ -122,6 +142,19 @@ extern "C" void GX__SetVtxAttrFmt_8016dc68(uint32_t vf, uint32_t a, uint32_t c,
}
GXSetVtxAttrFmt((GXVtxFmt)vf, (GXAttr)a, (GXCompCnt)c, (GXCompType)t, (u8)fr);
}
extern "C" void GX__SetVtxAttrFmt_8016dc68(uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr) {
const uint32_t attr = CanonicalVtxAttr(a);
if (vf < 8 && attr < 26) {
VtxAttrFmt& fmt = g_gxGameVertexState.vtxAttrFmt[vf][attr];
fmt.cnt = (GXCompCnt)c;
fmt.type = (GXCompType)t;
fmt.frac = (u8)fr;
if (a == GX_VA_NBT) {
g_gxGameVertexState.vtxAttrFmt[vf][GX_VA_NBT] = {};
}
}
GxThread::Post(&GX__SetVtxAttrFmt_8016dc68_gx, vf, a, c, t, fr);
}
PPC_NATIVE_OVERRIDE_VOID(8016dc68, GX__SetVtxAttrFmt_8016dc68, (uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr), (vf, a, c, t, fr));
extern "C" void GX__SetVtxAttrFmtv_8016de08(uint32_t vf, uint32_t la) {
@@ -139,7 +172,7 @@ extern "C" void GX__GetVtxAttrFmt_8016e04c(uint32_t vf, uint32_t a, uint32_t cp,
VtxAttrFmt fmt{};
const uint32_t attr = CanonicalVtxAttr(a);
if (vf < 8 && attr < GX_VA_MAX_ATTR && attr != GX_VA_NULL) {
fmt = g_hleGxState.vtxAttrFmt[vf][attr];
fmt = g_gxGameVertexState.vtxAttrFmt[vf][attr];
}
if (cp) {
Memory::Write32(cp, static_cast<uint32_t>(fmt.cnt));
@@ -162,7 +195,7 @@ extern "C" void GX__GetVtxAttrFmtv_8016e2b8(uint32_t vf, uint32_t la) {
for (uint32_t a = GX_VA_POS; a <= GX_VA_TEX7; ++a) {
VtxAttrFmt fmt{};
if (vf < 8) {
fmt = g_hleGxState.vtxAttrFmt[vf][a];
fmt = g_gxGameVertexState.vtxAttrFmt[vf][a];
}
Memory::Write32(p, a);
Memory::Write32(p + 4, static_cast<uint32_t>(fmt.cnt));
@@ -178,11 +211,11 @@ PPC_NATIVE_OVERRIDE_VOID(8016e2b8, GX__GetVtxAttrFmtv_8016e2b8, (uint32_t vf, ui
// Vertex Arrays
// ============================================================================
extern "C" void GX__SetArray_8016e32c(uint32_t a, uint32_t ba, uint32_t str) {
static void GX__SetArray_8016e32c_gx(uint32_t a, uint32_t ba, uint32_t str) {
const uint32_t attr = CanonicalVtxAttr(a);
if(attr<26){ g_hleGxState.vtxArray[attr].base=ba; g_hleGxState.vtxArray[attr].stride=str; }
}
PPC_NATIVE_OVERRIDE_VOID(8016e32c, GX__SetArray_8016e32c, (uint32_t a, uint32_t ba, uint32_t str), (a, ba, str));
GX_DEFERRED_OVERRIDE_VOID(8016e32c, GX__SetArray_8016e32c, (uint32_t a, uint32_t ba, uint32_t str), (a, ba, str));
// Switch-artifact entry point for the same SDK function; forwards rather than
// repeating the body.
@@ -193,12 +226,12 @@ PPC_NATIVE_OVERRIDE_VOID(8016e1c4, GX__SetArray_8016e1c4, (uint32_t a, uint32_t
// Texture Coordinate Generation
// ============================================================================
extern "C" void GX__SetNumTexGens_8016e5a4(uint32_t n) {
static void GX__SetNumTexGens_8016e5a4_gx(uint32_t n) {
GXSetNumTexGens((u8)n);
}
PPC_NATIVE_OVERRIDE_VOID(8016e5a4, GX__SetNumTexGens_8016e5a4, (uint32_t n), (n));
GX_DEFERRED_OVERRIDE_VOID(8016e5a4, GX__SetNumTexGens_8016e5a4, (uint32_t n), (n));
extern "C" void GX__SetTexCoordGen2_8016e37c(uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm) {
static void GX__SetTexCoordGen2_8016e37c_gx(uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm) {
if (sp >= static_cast<uint32_t>(GX_MAX_TEXGENSRC)) {
uint32_t pc = 0;
uint32_t lr = 0;
@@ -212,24 +245,24 @@ extern "C" void GX__SetTexCoordGen2_8016e37c(uint32_t dc, uint32_t f, uint32_t s
}
GXSetTexCoordGen2((GXTexCoordID)dc, (GXTexGenType)f, (GXTexGenSrc)sp, m, (GXBool)n, pm);
}
PPC_NATIVE_OVERRIDE_VOID(8016e37c, GX__SetTexCoordGen2_8016e37c, (uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm), (dc, f, sp, m, n, pm));
GX_DEFERRED_OVERRIDE_VOID(8016e37c, GX__SetTexCoordGen2_8016e37c, (uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm), (dc, f, sp, m, n, pm));
extern "C" void GX__EnableTexOffsets_8016f37c(uint32_t coord, uint32_t lineEnable, uint32_t pointEnable) {
static void GX__EnableTexOffsets_8016f37c_gx(uint32_t coord, uint32_t lineEnable, uint32_t pointEnable) {
GXEnableTexOffsets(static_cast<GXTexCoordID>(coord),
lineEnable ? GX_TRUE : GX_FALSE,
pointEnable ? GX_TRUE : GX_FALSE);
}
PPC_NATIVE_OVERRIDE_VOID(8016f37c, GX__EnableTexOffsets_8016f37c, (uint32_t coord, uint32_t lineEnable, uint32_t pointEnable), (coord, lineEnable, pointEnable));
GX_DEFERRED_OVERRIDE_VOID(8016f37c, GX__EnableTexOffsets_8016f37c, (uint32_t coord, uint32_t lineEnable, uint32_t pointEnable), (coord, lineEnable, pointEnable));
extern "C" void GX__SetLineWidth_8016f314(uint32_t width, uint32_t texOffsets) {
static void GX__SetLineWidth_8016f314_gx(uint32_t width, uint32_t texOffsets) {
GXSetLineWidth(static_cast<u8>(width), static_cast<GXTexOffset>(texOffsets));
}
PPC_NATIVE_OVERRIDE_VOID(8016f314, GX__SetLineWidth_8016f314, (uint32_t width, uint32_t texOffsets), (width, texOffsets));
GX_DEFERRED_OVERRIDE_VOID(8016f314, GX__SetLineWidth_8016f314, (uint32_t width, uint32_t texOffsets), (width, texOffsets));
extern "C" void GX__SetPointSize_8016f348(uint32_t pointSize, uint32_t texOffsets) {
static void GX__SetPointSize_8016f348_gx(uint32_t pointSize, uint32_t texOffsets) {
GXSetPointSize(static_cast<u8>(pointSize), static_cast<GXTexOffset>(texOffsets));
}
PPC_NATIVE_OVERRIDE_VOID(8016f348, GX__SetPointSize_8016f348, (uint32_t pointSize, uint32_t texOffsets), (pointSize, texOffsets));
GX_DEFERRED_OVERRIDE_VOID(8016f348, GX__SetPointSize_8016f348, (uint32_t pointSize, uint32_t texOffsets), (pointSize, texOffsets));
// ============================================================================
// Begin/End Drawing
@@ -284,6 +317,13 @@ void ServiceDeferredTimingDuringGxWork() {
}
}
void GX__Begin_gx(uint32_t t, uint32_t vf, uint32_t nv) {
ApplyAuroraVtxStateForBegin(static_cast<GXVtxFmt>(vf));
g_hleGxState.currentVtxFmt=(GXVtxFmt)vf; g_hleGxState.currentPrim=(GXPrimitive)t;
g_hleGxState.vertsRemaining=nv; g_hleGxState.inBegin=true; g_hleGxState.auroraBeginCalled=false;
g_hleGxState.fifoReadOffset = 0;
g_hleGxState.fifoByteCount=0; g_hleGxState.ResetVertex();
}
extern "C" void GX__Begin_8016f0f0(uint32_t t, uint32_t vf, uint32_t nv) {
if(IsDisplayListActive()){
WriteDisplayListData((u8)(t|vf), 1);
@@ -291,21 +331,19 @@ extern "C" void GX__Begin_8016f0f0(uint32_t t, uint32_t vf, uint32_t nv) {
return;
}
ServiceDeferredTimingDuringGxWork();
ApplyAuroraVtxStateForBegin(static_cast<GXVtxFmt>(vf));
g_hleGxState.currentVtxFmt=(GXVtxFmt)vf; g_hleGxState.currentPrim=(GXPrimitive)t;
g_hleGxState.vertsRemaining=nv; g_hleGxState.inBegin=true; g_hleGxState.auroraBeginCalled=false;
g_hleGxState.fifoReadOffset = 0;
g_hleGxState.fifoByteCount=0; g_hleGxState.ResetVertex();
GxThread::Post(&GX__Begin_gx, t, vf, nv);
}
PPC_NATIVE_OVERRIDE_VOID(8016f0f0, GX__Begin_8016f0f0, (uint32_t t, uint32_t vf, uint32_t nv), (t, vf, nv));
extern "C" void GX__End_80044b30() { g_hleGxState.inBegin=false; GXEnd(); }
PPC_NATIVE_OVERRIDE_VOID(80044b30, GX__End_80044b30, (), ());
static void GX__End_80044b30_gx() { g_hleGxState.inBegin=false; GXEnd(); }
GX_DEFERRED_OVERRIDE_VOID(80044b30, GX__End_80044b30, (), ());
extern "C" void GX__End_80048c30() { GX__End_80044b30(); }
PPC_NATIVE_OVERRIDE_VOID(80048c30, GX__End_80048c30, (), ());
extern "C" void GX__DrawSphere_80172a30(uint32_t numMajor, uint32_t numMinor) {
// Runs whole on the GX thread: it edits the parser's descriptor state and
// restores it, so the game-side mirror is unchanged on return.
static void GX__DrawSphere_80172a30_gx(uint32_t numMajor, uint32_t numMinor) {
constexpr uint32_t kAttrCount = 26;
constexpr float kSphereRadius = 1.0f;
constexpr float kPi = 3.14159265358979323846f;
@@ -330,13 +368,13 @@ extern "C" void GX__DrawSphere_80172a30(uint32_t numMajor, uint32_t numMinor) {
g_hleGxState.InvalidateVtxLayoutHash();
GXClearVtxDesc();
GX__SetVtxDesc_8016d3a4(GX_VA_POS, GX_DIRECT);
GX__SetVtxDesc_8016d3a4(GX_VA_NRM, GX_DIRECT);
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, GX_VA_POS, GX_POS_XYZ, GX_F32, 0);
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, GX_VA_NRM, GX_NRM_XYZ, GX_F32, 0);
GX__SetVtxDesc_8016d3a4_gx(GX_VA_POS, GX_DIRECT);
GX__SetVtxDesc_8016d3a4_gx(GX_VA_NRM, GX_DIRECT);
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, GX_VA_POS, GX_POS_XYZ, GX_F32, 0);
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, GX_VA_NRM, GX_NRM_XYZ, GX_F32, 0);
if (hadTex0) {
GX__SetVtxDesc_8016d3a4(GX_VA_TEX0, GX_DIRECT);
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, GX_VA_TEX0, GX_TEX_ST, GX_F32, 0);
GX__SetVtxDesc_8016d3a4_gx(GX_VA_TEX0, GX_DIRECT);
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, GX_VA_TEX0, GX_TEX_ST, GX_F32, 0);
}
const float majorStep = kPi / static_cast<float>(numMajor);
@@ -400,12 +438,12 @@ extern "C" void GX__DrawSphere_80172a30(uint32_t numMajor, uint32_t numMinor) {
for (uint32_t attr = 0; attr < kAttrCount; ++attr) {
const GXAttrType type = savedVtxDesc[attr];
if (type != GX_NONE) {
GX__SetVtxDesc_8016d3a4(attr, static_cast<uint32_t>(type));
GX__SetVtxDesc_8016d3a4_gx(attr, static_cast<uint32_t>(type));
}
}
for (uint32_t attr = GX_VA_POS; attr <= GX_VA_TEX7; ++attr) {
const VtxAttrFmt& fmt = savedFmt3[attr];
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, attr, static_cast<uint32_t>(fmt.cnt), static_cast<uint32_t>(fmt.type), fmt.frac);
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, attr, static_cast<uint32_t>(fmt.cnt), static_cast<uint32_t>(fmt.type), fmt.frac);
}
}
PPC_NATIVE_OVERRIDE_VOID(80172a30, GX__DrawSphere_80172a30, (uint32_t numMajor, uint32_t numMinor), (numMajor, numMinor));
GX_DEFERRED_OVERRIDE_VOID(80172a30, GX__DrawSphere_80172a30, (uint32_t numMajor, uint32_t numMinor), (numMajor, numMinor));
+83 -25
View File
@@ -4,6 +4,10 @@
#include "guest_interrupt_context.h"
#include "ppc_runtime.h"
#include "aurora_events.h"
#include "gx_thread.h"
#include <aurora/imgui.h>
#include <algorithm>
#include <array>
#include "settings_overlay.h"
#include "fiber_manager.h"
#include "platform/host_platform.h"
@@ -48,6 +52,16 @@ extern "C" int32_t OS__RestoreInterrupts_801a65d4(int32_t level);
std::atomic_bool g_auroraFrameActive{false};
std::atomic_bool g_auroraFrameHadWork{false};
// GX-thread routine (inline when the thread is off): the aurora frame flags are
// written only where aurora's producer runs.
static void GxBeginAuroraFrameIfIdle_gx() {
if (!g_auroraFrameActive.load(std::memory_order_acquire)) {
if (BeginAuroraFrame()) {
g_auroraFrameActive.store(true, std::memory_order_release);
}
}
}
namespace {
using namespace std::chrono_literals;
@@ -328,12 +342,10 @@ void AdvanceRetrace(CpuContext* ctx, Clock::time_point retraceStamp, bool servic
// Process window events (but don't present - that happens in GXCopyDisp)
UpdateAuroraAndProcessEvents();
// Start a new Aurora frame if one isn't already active
if (!g_auroraFrameActive.load(std::memory_order_acquire)) {
if (BeginAuroraFrame()) {
g_auroraFrameActive.store(true, std::memory_order_release);
}
}
// Start a new Aurora frame if one isn't already active. The attempt runs
// where aurora's producer runs; the viewport policy stays on this thread.
GxThread::Post(&GxBeginAuroraFrameIfIdle_gx);
ApplyPendingMkwDynamicAspectSurface();
// XR runtime loss and encoded-work stalls request producer-owned
// teardown. Service it on every retrace, including startup, pause, and
// minimized-window paths that may never reach a successful present.
@@ -522,24 +534,65 @@ void PaceToRetraceBoundary(Clock::time_point deadline) {
// pacing thread: the anchor only makes sense against the recorded camera of
// this exact frame, and the pacing thread does not know which frame its packet
// will be paired with.
void PublishVrSceneAnchor() {
struct SceneAnchorPublication {
std::array<float, 12> anchor{};
bool valid = false;
};
SceneAnchorPublication PublishVrSceneAnchor() {
static bool s_engaged = false;
// Compute the anchor here rather than at the race draw boundary: the
// scene's camera matrix for this frame is only set once the draws run, so
// reading it earlier pairs a stale camera with a current kart pose.
mkw::vr::MkwVRFirstPersonCommit();
const mkw::vr::FirstPersonAnchor anchor = mkw::vr::MkwVRFirstPersonGetAnchor();
aurora_set_stereo_scene_anchor(anchor.valid ? anchor.anchor_from_scene.data() : nullptr);
SceneAnchorPublication publication;
publication.valid = anchor.valid;
if (anchor.valid) {
std::copy(anchor.anchor_from_scene.begin(), anchor.anchor_from_scene.end(), publication.anchor.begin());
}
const bool engaged = anchor.valid;
if (engaged == s_engaged) {
return;
return publication;
}
s_engaged = engaged;
// The world scale changes with the camera, and the virtual screen's metres
// are converted at that scale, so the two have to move together.
mkw::vr::MkwVRPolicySetFirstPersonEngaged(engaged);
settings_overlay::RefreshVrHudVirtualScreen();
return publication;
}
// Everything the seal needs, latched on the game thread and applied by the GX
// thread in one record, so the schedule, anchor, tag and overlay all belong to
// the frame whose GX commands precede it in the ring.
struct GxPresentRecord {
uint64_t scheduleBaseNanos = 0;
uint64_t scheduleIntervalNanos = 0;
std::array<float, 12> anchor{};
bool anchorValid = false;
bool reportPaced = false;
bool paced = false;
uint32_t localPlayerCount = 1;
uint64_t contentTag = 0;
void* imguiFrame = nullptr;
};
void GxPresent_gx(GxPresentRecord record) {
if (record.reportPaced) {
aurora_report_producer_paced(record.paced);
}
aurora_set_present_schedule(record.scheduleBaseNanos, record.scheduleIntervalNanos);
aurora_set_stereo_scene_anchor(record.anchorValid ? record.anchor.data() : nullptr);
aurora_set_stereo_local_player_count(record.localPlayerCount);
aurora_end_frame_ex(record.contentTag, record.imguiFrame);
g_auroraFrameActive.store(false, std::memory_order_release);
g_auroraFrameHadWork.store(false, std::memory_order_release);
// Pre-warm the next frame so subsequent GX work has a valid frame context.
if (BeginAuroraFrame()) {
g_auroraFrameActive.store(true, std::memory_order_release);
}
}
} // namespace
@@ -556,6 +609,7 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) {
} sequenceGuard;
Clock::time_point paceDeadline{};
bool paceThisFrame = false;
GxPresentRecord record;
if (paceToRetrace) {
uint64_t baseNanos = 0;
uint64_t intervalNanos = 0;
@@ -586,7 +640,8 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) {
// cadence, and zero only happens when production outruns VI. "Kept up" means <=1 retrace
// elapsed; 2+ means a boundary was missed, so Aurora seals that frame without its interpolated
// slots (a windowed backstop lowers the slot target only under sustained overload).
aurora_report_producer_paced(retracesElapsed <= 1);
record.reportPaced = true;
record.paced = retracesElapsed <= 1;
// Stamp the sealed frame's presentation schedule so Aurora paces interpolated slots against
// this same VI timeline. Anchor to the NEXT retrace boundary, not the period just produced,
// since slots anchored to the current period would already be expired by seal time. Encoding
@@ -599,47 +654,50 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) {
anchorNanos = s_lastPresentAnchorNanos + intervalNanos;
}
s_lastPresentAnchorNanos = anchorNanos;
aurora_set_present_schedule(anchorNanos, intervalNanos);
record.scheduleBaseNanos = anchorNanos;
record.scheduleIntervalNanos = intervalNanos;
} else {
// Retrace-context presents (VI black, boot) have no display period of
// their own to subdivide; present as soon as the frame is ready. The
// schedule grid is gone, so the anchor cursor must not constrain the
// next paced frame.
s_lastPresentAnchorNanos = 0;
aurora_set_present_schedule(0, 0);
}
mkw::vr::OpenXRServiceProducerFrameBoundary();
PublishVrSceneAnchor();
const SceneAnchorPublication anchor = PublishVrSceneAnchor();
record.anchor = anchor.anchor;
record.anchorValid = anchor.valid;
// Latch the current policy safety state into this exact Aurora job. The
// asynchronous worker may ask for an XR packet after the guest has already
// begun the next frame, so immersive replay is accepted only when both
// tags match.
const auto vrPolicy = mkw::vr::MkwVRPolicyGetSnapshot();
aurora_set_stereo_local_player_count(
record.localPlayerCount =
vrPolicy.presentation == mkw::vr::VRPresentationMode::ImmersiveRace
? vrPolicy.scene.local_player_count : 1);
aurora_end_frame_tagged(vrPolicy.content_tag);
? vrPolicy.scene.local_player_count : 1;
record.contentTag = vrPolicy.content_tag;
// The overlay drew into the game thread's ImGui frame; close it and hand a
// copy of its draw data to the seal.
record.imguiFrame = aurora_imgui_host_frame_end();
GxThread::Post(&GxPresent_gx, record);
if (paceThisFrame) {
PaceToRetraceBoundary(paceDeadline);
std::lock_guard<std::mutex> lock(g_viMutex);
s_lastPacedRetraceCount = g_vi.retraceCount;
}
settings_overlay::AdvancePresentedFrame();
g_auroraFrameActive.store(false, std::memory_order_release);
g_auroraFrameHadWork.store(false, std::memory_order_release);
if (presentedXfb) {
std::lock_guard<std::mutex> lock(g_viMutex);
g_vi.hasValidXfb = false;
g_vi.readyXfb = 0;
}
// Pre-warm the next frame so subsequent GX work has a valid frame context.
{
UpdateAuroraAndProcessEvents();
if (BeginAuroraFrame()) {
g_auroraFrameActive.store(true, std::memory_order_release);
}
}
// The GX thread begins the next frame right after the seal; the window's
// thread pumps events, applies any surface change and starts the next
// ImGui frame for the overlay.
UpdateAuroraAndProcessEvents();
ApplyPendingMkwDynamicAspectSurface();
aurora_imgui_host_frame_begin();
}
// -----------------------------------------------------------------------------
+26
View File
@@ -54,6 +54,8 @@
#include "abi_bridge.h"
#include "guest_flat_memory.h"
#include "gx_guest_write.h"
#include "gx_thread.h"
#include <aurora/imgui.h>
#include "memory.h"
#include "system_bridge.h"
#include "ppc_runtime.h"
@@ -99,6 +101,14 @@ void ServiceGuestTimingDuringAuroraFrameWait() {
Audio_HLE_PollDeferred();
}
void GxThreadFrameLog(char* buffer, uint32_t bufferSize, double windowSeconds, uint32_t frames) {
if (!GxThread::Enabled()) {
return;
}
const std::string line = GxThread::FormatStatsAndReset(windowSeconds, frames);
std::snprintf(buffer, bufferSize, "%s", line.c_str());
}
#if defined(_WIN32)
int __cdecl WindowsCrtReportHook(int reportType, char* message, int* returnValue) {
if (returnValue) {
@@ -1497,6 +1507,19 @@ int RuntimeMain(int argc, char** argv) {
}
aurora_set_frame_worker_wait_callback(ServiceGuestTimingDuringAuroraFrameWait);
GxGuestWrite::InstallAuroraHooks();
GxThread::Configure(RuntimeConfigFile::GxThread());
GxThread::SetWaitCallback(ServiceGuestTimingDuringAuroraFrameWait);
GxThread::Start();
if (GxThread::Enabled()) {
// The GX thread is aurora's producer: the game thread never waits
// inside aurora any more, and it pumps SDL itself (aurora_update).
aurora_set_frame_worker_wait_callback(nullptr);
aurora_set_host_event_pump(true);
}
aurora_set_frame_log_callback(GxThreadFrameLog);
// The desktop overlay's ImGui frames belong to this thread from the
// first frame on; the seal replays a copy of their draw data.
aurora_imgui_host_frame_begin();
UpdateMkwDynamicAspectSurface(auroraInfo.windowSize.native_fb_width,
auroraInfo.windowSize.native_fb_height);
settings_overlay::InitializeRuntimeSettings();
@@ -1549,6 +1572,7 @@ int RuntimeMain(int argc, char** argv) {
// Shutdown fiber system
Fiber::GuestFiberManager::Shutdown();
WindowPlacementPersistence::Flush(true);
GxThread::Stop();
mkw::vr::OpenXRShutdownBeforeAurora();
aurora_shutdown();
DiscordPresence::Shutdown();
@@ -1568,6 +1592,7 @@ int RuntimeMain(int argc, char** argv) {
SetRuntimeExitCodeImpl(1);
Fiber::GuestFiberManager::Shutdown();
WindowPlacementPersistence::Flush(true);
GxThread::Stop();
mkw::vr::OpenXRShutdownBeforeAurora();
aurora_shutdown();
DiscordPresence::Shutdown();
@@ -1581,6 +1606,7 @@ int RuntimeMain(int argc, char** argv) {
SetRuntimeExitCodeImpl(1);
Fiber::GuestFiberManager::Shutdown();
WindowPlacementPersistence::Flush(true);
GxThread::Stop();
mkw::vr::OpenXRShutdownBeforeAurora();
aurora_shutdown();
DiscordPresence::Shutdown();
+3 -3
View File
@@ -2143,9 +2143,9 @@ void ReleaseControllers() noexcept {
}
void Draw() noexcept {
// Wait for the frame worker's DONE phase: it has replayed the previous frame's ImGui draw lists
// and started the next ImGui frame, so all overlay callers can now safely issue ImGui commands.
aurora_wait_for_frame_worker();
// The overlay draws into the game thread's own ImGui frame
// (aurora_imgui_host_frame_begin); the worker replays a copy of the draw
// data, so there is nothing to wait for here.
// Explain fallback without interrupting gameplay or capturing input.
static std::string shownXrError;
static double xrNoticeUntil = 0.0;
+19
View File
@@ -7,6 +7,7 @@
#include "vr/openxr_integration.h"
#include "runtime_config.h"
#include "gx_thread.h"
#include "runtime_log.h"
#include "vr/mkw_vr_first_person.h"
#include "vr/mkw_vr_policy.h"
@@ -473,6 +474,9 @@ public:
void Shutdown() noexcept {
teardown_requested_.store(false, std::memory_order_release);
// Called on the game thread: aurora's producer (the GX thread) must be
// idle before the frame worker is quiesced.
GxThread::Drain();
// Stop idle replays before draining; no new worker job may race provider removal.
{
std::lock_guard lock(interpolation_mutex_);
@@ -637,6 +641,16 @@ private:
<< (hinted ? "set" : "refused") << std::endl;
return true;
}
bool RegisterGxThread() {
const uint32_t thread_id = GxThread::NativeThreadId();
if (thread_id == 0 || runtime_ == nullptr) {
return false;
}
const bool hinted = OpenXRAndroidRegisterThreadId(*runtime_, OpenXRAndroidThreadType::RendererWorker, thread_id);
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: Android thread hint for the GX thread "
<< (hinted ? "set" : "refused") << std::endl;
return true;
}
#endif
// Asks the runtime for the configured performance level in both domains. Standalone
@@ -689,6 +703,7 @@ private:
// frame worker, which submits the GPU work, are the ones that matter; this thread only
// paces.
bool worker_registered = false;
bool gx_registered = false;
if (runtime_ != nullptr) {
const bool pacing_hinted =
OpenXRAndroidRegisterThread(*runtime_, OpenXRAndroidThreadType::RendererWorker);
@@ -700,6 +715,7 @@ private:
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: Android thread hints: game " << (game_hinted ? "set" : "refused")
<< ", pacing " << (pacing_hinted ? "set" : "refused") << std::endl;
worker_registered = RegisterAuroraFrameWorkerThread();
gx_registered = RegisterGxThread();
}
#endif
ApplyPerformanceLevel();
@@ -716,6 +732,9 @@ private:
if (!worker_registered) {
worker_registered = RegisterAuroraFrameWorkerThread();
}
if (!gx_registered) {
gx_registered = RegisterGxThread();
}
#endif
const OpenXREventStatus events = runtime_->PollEvents();
const bool session_active = runtime_->IsSessionRunning();