mirror of
https://github.com/mitch030504/Wiicompiled_VR_Frame.git
synced 2026-10-06 05:00:27 +02:00
Implement GX Thread for Asynchronous Rendering
- Introduced a new GX thread to handle the rendering pipeline, allowing the game thread to post commands without blocking. - Added `gx_thread.h` and `gx_thread.cpp` to manage the command ring buffer and thread synchronization. - Updated `vi.cpp` to utilize the GX thread for rendering tasks, improving frame pacing and responsiveness. - Modified `main.cpp` to configure and start the GX thread, ensuring it integrates with the existing rendering workflow. - Enhanced `openxr_integration.cpp` to register the GX thread with OpenXR for better performance in VR scenarios. - Refactored `settings_overlay.cpp` to remove unnecessary waits for the frame worker, as the overlay now draws directly into the game thread's ImGui frame. - Improved error handling and logging in the GX thread to capture exceptions during command execution.
This commit is contained in:
1 parent
ffa63fa60a
commit
6a1641e0b7
33 files changed
+1532
-527
No files matched your search
@@ -78,7 +78,7 @@ object GameStorage {
|
||||
// Written line by line: trimIndent runs after interpolation, so an interpolated line
|
||||
// would take the indent off every other one.
|
||||
val lines = mutableListOf(
|
||||
"# WiiCompiled Quest configuration. Edit with the launcher, the in-game panel or adb pull/push.",
|
||||
"# WiiCompiled Quest configuration. Edit with the launcher or the in-game panel. After an adb push, run chmod 664 on it or the app can no longer save settings.",
|
||||
"[paths]",
|
||||
"dvd_root = \"${discDirectory(context).absolutePath}\"",
|
||||
)
|
||||
|
||||
@@ -227,6 +227,11 @@ class SettingsPage(
|
||||
read = { it.bool("video", "skip_unready_pipelines") ?: true },
|
||||
write = { c, value -> c.setBool("video", "skip_unready_pipelines", value) },
|
||||
)
|
||||
toggle(
|
||||
R.string.graphics_gx_thread, R.string.graphics_gx_thread_helper,
|
||||
read = { it.bool("video", "gx_thread") ?: true },
|
||||
write = { c, value -> c.setBool("video", "gx_thread", value) },
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -188,6 +188,8 @@
|
||||
<string name="graphics_bloom_helper">Bloom\'s bright glow reads poorly in a headset, so it starts off.</string>
|
||||
<string name="graphics_skip_unready">Prevent shader stutters</string>
|
||||
<string name="graphics_skip_unready_helper">Skips a draw for a moment while its shader compiles instead of pausing the game.</string>
|
||||
<string name="graphics_gx_thread">Graphics thread</string>
|
||||
<string name="graphics_gx_thread_helper">Prepares the drawing on a second CPU core so busy scenes keep their speed. Turn off only to compare against the single-threaded path. Takes effect on the next launch.</string>
|
||||
|
||||
<!-- Controls tab -->
|
||||
<string name="section_controls">Controllers</string>
|
||||
|
||||
@@ -218,6 +218,17 @@ void aurora_end_frame();
|
||||
// Seal the current frame with an opaque application safety tag. Aurora rejects
|
||||
// an immersive provider packet unless its contentTag matches this exact frame.
|
||||
void aurora_end_frame_tagged(uint64_t contentTag);
|
||||
// aurora_end_frame_tagged() plus the host-owned ImGui frame to present with it (the handle from
|
||||
// aurora_imgui_host_frame_end(), which this call consumes; NULL presents no host ImGui frame).
|
||||
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame);
|
||||
// When the host pumps SDL events itself (aurora_update() on the window's thread) and drives
|
||||
// begin/end frame from another thread, this stops those calls from pumping events.
|
||||
void aurora_set_host_event_pump(bool hostPumps);
|
||||
typedef void (*AuroraFrameLogCallback)(char* buffer, uint32_t bufferSize, double windowSeconds,
|
||||
uint32_t frames);
|
||||
// Called with each five-second frame-rate log window (where that log is enabled); a non-empty
|
||||
// buffer is logged as one extra line.
|
||||
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback);
|
||||
/**
|
||||
* Relocates the immersive camera for the frame about to be sealed.
|
||||
*
|
||||
|
||||
@@ -25,6 +25,14 @@ ImTextureID aurora_imgui_add_texture(uint32_t width, uint32_t height, const void
|
||||
// producer before the frame is sealed, and leave the draw data untouched until the frame worker is
|
||||
// done with that frame (aurora_wait_for_frame_worker). Null hides the panel.
|
||||
void aurora_imgui_set_stereo_overlay(ImDrawData* drawData, float widthFraction);
|
||||
// Host-owned ImGui frames for the desktop overlay. Begin starts the next frame on the calling
|
||||
// thread (which must be the window's thread, since the SDL backend reads the window there); end
|
||||
// renders it and returns a handle to a private copy of its draw data, which aurora_end_frame_ex()
|
||||
// consumes. A handle that is never presented is freed with aurora_imgui_host_frame_release().
|
||||
// From the first begin on, aurora no longer starts ImGui frames itself.
|
||||
void aurora_imgui_host_frame_begin(void);
|
||||
void* aurora_imgui_host_frame_end(void);
|
||||
void aurora_imgui_host_frame_release(void* imguiFrame);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
|
||||
+84
-23
@@ -69,6 +69,9 @@ AuroraConfig g_config;
|
||||
uint32_t g_sdlCustomEventsStart;
|
||||
char g_gameName[4];
|
||||
std::atomic<AuroraFrameWorkerWaitCallback> g_frameWorkerWaitCallback{nullptr};
|
||||
// aurora_set_host_event_pump(): the host pumps SDL itself, from the window's thread.
|
||||
std::atomic_bool g_hostEventPump{false};
|
||||
std::atomic<AuroraFrameLogCallback> g_frameLogCallback{nullptr};
|
||||
// Presentation schedule for the frame being sealed, set by the producer. Jobs carry absolute
|
||||
// deadlines derived from it, so the presenter cannot drift. Zero means present when ready.
|
||||
std::atomic<uint64_t> g_presentScheduleBaseNanos{0};
|
||||
@@ -266,7 +269,8 @@ enum class ImGuiFramePolicy {
|
||||
bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy = ImGuiFramePolicy::Immediate,
|
||||
bool* imguiNewFrameOwed = nullptr) noexcept;
|
||||
bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewFrameOwed) noexcept;
|
||||
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor) noexcept;
|
||||
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag, const StereoSceneAnchor& sceneAnchor,
|
||||
imgui::HostFramePtr hostImGuiFrame) noexcept;
|
||||
|
||||
// The two publication points of a frame-worker cycle, cleared together under `mutex`. Sealed:
|
||||
// producer-shared renderer state is free again. Done: slots encoded, presented, ImGui restarted.
|
||||
@@ -288,6 +292,7 @@ struct FrameWorkerState {
|
||||
// belong to that exact queued frame, not to the producer's next frame.
|
||||
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
|
||||
StereoSceneAnchor sceneAnchor{};
|
||||
imgui::HostFramePtr hostImGuiFrame;
|
||||
// Readiness is polled thousands of times per frame, so these flags double as a publication
|
||||
// barrier. `sealed` is released before `ready`, and both are cleared under `mutex`.
|
||||
std::atomic_bool sealed{true};
|
||||
@@ -338,7 +343,7 @@ bool frame_worker_requested() noexcept {
|
||||
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
// Returns false when a stop request was observed mid-cycle.
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
|
||||
const StereoSceneAnchor& sceneAnchor) noexcept;
|
||||
void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept;
|
||||
#endif
|
||||
@@ -361,6 +366,7 @@ void frame_worker_main() noexcept {
|
||||
for (;;) {
|
||||
uint64_t contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
|
||||
StereoSceneAnchor sceneAnchor{};
|
||||
imgui::HostFramePtr hostImGuiFrame;
|
||||
bool stereoOnly = false;
|
||||
{
|
||||
std::unique_lock lock(g_frameWorker.mutex);
|
||||
@@ -377,6 +383,7 @@ void frame_worker_main() noexcept {
|
||||
g_frameWorker.contentTag = AURORA_STEREO_CONTENT_TAG_UNKNOWN;
|
||||
sceneAnchor = g_frameWorker.sceneAnchor;
|
||||
g_frameWorker.sceneAnchor = {};
|
||||
hostImGuiFrame = std::move(g_frameWorker.hostImGuiFrame);
|
||||
g_frameWorker.jobPending = false;
|
||||
}
|
||||
|
||||
@@ -396,7 +403,7 @@ void frame_worker_main() noexcept {
|
||||
g_frameWorker.cv.notify_all();
|
||||
continue;
|
||||
}
|
||||
if (!run_frame_worker_cycle(sealedFrame, contentTag, sceneAnchor)) {
|
||||
if (!run_frame_worker_cycle(sealedFrame, contentTag, std::move(hostImGuiFrame), sceneAnchor)) {
|
||||
break;
|
||||
}
|
||||
#else
|
||||
@@ -1556,7 +1563,7 @@ void publish_stereo_screen_aspects(const webgpu::PresentSource& presentSource, c
|
||||
// gfx::begin_frame() may already have cleared the display-copy override.
|
||||
void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const webgpu::PresentSource& presentSource,
|
||||
const PresentationImage& image, bool includeImGui,
|
||||
MirrorPlan plan = MirrorPlan::Mono) {
|
||||
MirrorPlan plan = MirrorPlan::Mono, const ImDrawData* hostImGuiData = nullptr) {
|
||||
ZoneScoped;
|
||||
auto viewport = webgpu::calculate_present_viewport(image.texture.size.width, image.texture.size.height,
|
||||
presentSource.size.width, presentSource.size.height);
|
||||
@@ -1635,7 +1642,11 @@ void encode_presentation_snapshot(const wgpu::CommandEncoder& encoder, const web
|
||||
const auto pass = encoder.BeginRenderPass(&renderPassDescriptor);
|
||||
pass.SetViewport(0.f, 0.f, static_cast<float>(image.texture.size.width),
|
||||
static_cast<float>(image.texture.size.height), 0.f, 1.f);
|
||||
imgui::render(pass);
|
||||
if (hostImGuiData != nullptr) {
|
||||
imgui::render(pass, hostImGuiData);
|
||||
} else {
|
||||
imgui::render(pass);
|
||||
}
|
||||
pass.End();
|
||||
}
|
||||
}
|
||||
@@ -1678,7 +1689,7 @@ bool begin_frame_impl(bool pumpEvents, ImGuiFramePolicy imguiPolicy, bool* imgui
|
||||
ZoneScoped;
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
webgpu::fail_if_device_lost();
|
||||
if (pumpEvents) {
|
||||
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
|
||||
window::pump_events();
|
||||
}
|
||||
const bool surfaceReconfigurePending = g_surfaceReconfigurePending.load(std::memory_order_acquire);
|
||||
@@ -1735,7 +1746,9 @@ bool begin_frame_render_state_impl(ImGuiFramePolicy imguiPolicy, bool* imguiNewF
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
// Note the debt before gfx::begin_frame() can fail: the synchronous path always started the
|
||||
// ImGui frame here, and the runtime's retry loop depends on that pairing.
|
||||
if (imguiPolicy == ImGuiFramePolicy::Immediate) {
|
||||
if (imgui::host_frames_active()) {
|
||||
// The host starts its own ImGui frames (imgui::host_frame_begin).
|
||||
} else if (imguiPolicy == ImGuiFramePolicy::Immediate) {
|
||||
imgui::new_frame(window::get_window_size());
|
||||
} else if (imguiNewFrameOwed != nullptr) {
|
||||
*imguiNewFrameOwed = true;
|
||||
@@ -1771,8 +1784,13 @@ struct SealedFrameContext {
|
||||
std::optional<AuroraStereoFrame> stereoInput;
|
||||
bool retainStereo = false;
|
||||
imgui::StereoOverlay stereoOverlay;
|
||||
imgui::HostFramePtr imguiFrame;
|
||||
};
|
||||
|
||||
const ImDrawData* host_imgui_data(const SealedFrameContext& ctx) noexcept {
|
||||
return ctx.imguiFrame ? imgui::host_frame_draw_data(*ctx.imguiFrame) : nullptr;
|
||||
}
|
||||
|
||||
// Worker-owned scene state. A separate buffer generation check protects against
|
||||
// synchronous EFB submissions overwriting the retained frame's GPU data.
|
||||
struct RetainedStereoContext {
|
||||
@@ -1831,7 +1849,7 @@ void run_retained_stereo_frame(gfx::SealedFrame& sealedFrame) noexcept {
|
||||
// Phase 1: everything that touches producer-shared renderer state. Needs g_rendererGpuMutex and
|
||||
// a FIFO already drained into the recorded pass list.
|
||||
void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, uint64_t contentTag,
|
||||
const StereoSceneAnchor& sceneAnchor) {
|
||||
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) {
|
||||
ZoneScopedN("Seal frame");
|
||||
// Every pass this cycle encodes, from the seal's probe blits to the final eye, is timed under
|
||||
// one frame; encode_sealed_frame resolves it on its last submission.
|
||||
@@ -1889,7 +1907,12 @@ void seal_frame_locked(gfx::SealedFrame& sealedFrame, SealedFrameContext& ctx, u
|
||||
ctx.presentSource = webgpu::current_present_source();
|
||||
// ImGui draw lists are built once per frame and replayed by each slot's ImGui pass, which is why
|
||||
// the next ImGui frame cannot start until the encode phase is done.
|
||||
imgui::render_frame_data();
|
||||
if (hostImGuiFrame) {
|
||||
// The host closed its own ImGui frame and handed over a copy of the draw data.
|
||||
ctx.imguiFrame = std::move(hostImGuiFrame);
|
||||
} else {
|
||||
imgui::render_frame_data();
|
||||
}
|
||||
// The headset panel's draw data follows the same rule on the host's side.
|
||||
ctx.stereoOverlay = imgui::latch_stereo_overlay();
|
||||
// Drop the sealed frame's lazy RAM-readback requests while the producer is still excluded; it
|
||||
@@ -1981,7 +2004,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
|
||||
gfx::render(sealedFrame, encoder, static_cast<int32_t>(interpolatedFrame), false);
|
||||
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
|
||||
presentationJobs.push_back({
|
||||
.image = std::move(image),
|
||||
.logicalFrame = ctx.logicalFrame,
|
||||
@@ -2013,7 +2036,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
if (!ctx.replayInterpolatedFrames) {
|
||||
for (uint32_t interpolatedFrame = 0; interpolatedFrame < ctx.interpolatedFrameCount; ++interpolatedFrame) {
|
||||
auto image = acquire_presentation_image(interpolatedFrame, ctx.snapshotWidth, ctx.snapshotHeight);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *image, true, mirrorPlan, host_imgui_data(ctx));
|
||||
presentationJobs.push_back({
|
||||
.image = std::move(image),
|
||||
.logicalFrame = ctx.logicalFrame,
|
||||
@@ -2053,7 +2076,7 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
// showing. Black re-clears it below, once the eyes have taken their copy.
|
||||
const bool virtualScreenNeedsMono = stereoOutput && !immersiveReplay;
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true,
|
||||
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan);
|
||||
virtualScreenNeedsMono ? MirrorPlan::Mono : mirrorPlan, host_imgui_data(ctx));
|
||||
if (stereoOutput) {
|
||||
publish_stereo_screen_aspects(ctx.presentSource, finalImage->texture.size, immersiveReplay);
|
||||
}
|
||||
@@ -2075,7 +2098,8 @@ std::vector<PresentationJob> encode_sealed_frame(gfx::SealedFrame& sealedFrame,
|
||||
stereo_overlay::composite_flat(encoder, output.view, output.size, eye);
|
||||
}
|
||||
if (mirrorPlan == MirrorPlan::Black && !headsetOnly) {
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black);
|
||||
encode_presentation_snapshot(encoder, ctx.presentSource, *finalImage, true, MirrorPlan::Black,
|
||||
host_imgui_data(ctx));
|
||||
}
|
||||
}
|
||||
if (stereoOutput) {
|
||||
@@ -2231,6 +2255,14 @@ void record_frame_telemetry() {
|
||||
if (const std::string gpuTiming = gfx::gpu_timing_report(); !gpuTiming.empty()) {
|
||||
Log.info("{}", gpuTiming);
|
||||
}
|
||||
if (const auto frameLog = g_frameLogCallback.load(std::memory_order_acquire)) {
|
||||
char extra[512];
|
||||
extra[0] = '\0';
|
||||
frameLog(extra, sizeof(extra), elapsed.count(), windowFrames);
|
||||
if (extra[0] != '\0') {
|
||||
Log.info("{}", extra);
|
||||
}
|
||||
}
|
||||
windowStart = now;
|
||||
windowFrames = 0;
|
||||
}
|
||||
@@ -2242,7 +2274,7 @@ void record_frame_telemetry() {
|
||||
|
||||
// One complete frame-worker cycle. Desktop and headset interpolation both
|
||||
// release the producer after sealing, before encoding their extra scene views.
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame,
|
||||
const StereoSceneAnchor& sceneAnchor) noexcept {
|
||||
ZoneScopedN("Frame worker cycle");
|
||||
webgpu::fail_if_device_lost();
|
||||
@@ -2254,7 +2286,7 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
auto stretchStarted = std::chrono::steady_clock::now();
|
||||
{
|
||||
std::lock_guard gpuLock(g_rendererGpuMutex);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
|
||||
}
|
||||
g_workerSealNs.fetch_add(elapsedNs(stretchStarted), std::memory_order_relaxed);
|
||||
stretchStarted = std::chrono::steady_clock::now();
|
||||
@@ -2311,11 +2343,11 @@ bool run_frame_worker_cycle(gfx::SealedFrame& sealedFrame, uint64_t contentTag,
|
||||
// Synchronous frame submission: seal, encode and present inline on the calling thread. Used when
|
||||
// the frame worker is disabled (RenderDoc captures) and on the boot path.
|
||||
void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
|
||||
const StereoSceneAnchor& sceneAnchor) noexcept {
|
||||
const StereoSceneAnchor& sceneAnchor, imgui::HostFramePtr hostImGuiFrame) noexcept {
|
||||
ZoneScoped;
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
webgpu::fail_if_device_lost();
|
||||
if (pumpEvents) {
|
||||
if (pumpEvents && !g_hostEventPump.load(std::memory_order_acquire)) {
|
||||
window::pump_events();
|
||||
}
|
||||
gfx::SealedFrame sealedFrame;
|
||||
@@ -2326,7 +2358,7 @@ void end_frame_impl(bool pumpEvents, bool drainFifo, uint64_t contentTag,
|
||||
if (drainFifo) {
|
||||
gx::fifo::drain();
|
||||
}
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor);
|
||||
seal_frame_locked(sealedFrame, ctx, contentTag, sceneAnchor, std::move(hostImGuiFrame));
|
||||
presentationJobs = encode_sealed_frame(sealedFrame, ctx);
|
||||
}
|
||||
publish_presentations(std::move(presentationJobs), ctx.interpolationActive);
|
||||
@@ -2352,7 +2384,9 @@ bool begin_frame() noexcept {
|
||||
ensure_frame_worker_started();
|
||||
// SDL needs event pumping on the window-owning producer thread, and the worker passes
|
||||
// pumpEvents=false, so keep it here even when the fast path returns early.
|
||||
window::pump_events();
|
||||
if (!g_hostEventPump.load(std::memory_order_acquire)) {
|
||||
window::pump_events();
|
||||
}
|
||||
bool waitForSurfacePreparation = false;
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
// A surface mutation can legitimately fail preparation, and optimistic success would let GX/ImGui
|
||||
@@ -2402,7 +2436,7 @@ bool begin_frame() noexcept {
|
||||
return prepared;
|
||||
}
|
||||
|
||||
void end_frame(uint64_t contentTag) noexcept {
|
||||
void end_frame(uint64_t contentTag, imgui::HostFramePtr hostImGuiFrame) noexcept {
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
webgpu::fail_if_device_lost();
|
||||
#endif
|
||||
@@ -2413,7 +2447,7 @@ void end_frame(uint64_t contentTag) noexcept {
|
||||
g_pendingStereoLocalPlayerCount = 1;
|
||||
g_pendingSceneAnchor = {};
|
||||
if (!frame_worker_requested()) {
|
||||
end_frame_impl(true, true, contentTag, sceneAnchor);
|
||||
end_frame_impl(true, true, contentTag, sceneAnchor, std::move(hostImGuiFrame));
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2435,6 +2469,7 @@ void end_frame(uint64_t contentTag) noexcept {
|
||||
g_frameWorker.ready.store(false, std::memory_order_release);
|
||||
g_frameWorker.contentTag = contentTag;
|
||||
g_frameWorker.sceneAnchor = sceneAnchor;
|
||||
g_frameWorker.hostImGuiFrame = std::move(hostImGuiFrame);
|
||||
g_frameWorker.jobPending = true;
|
||||
g_frameWorker.prepareAllowed = false;
|
||||
}
|
||||
@@ -2537,8 +2572,34 @@ AuroraInfo aurora_initialize(int argc, char* argv[], const AuroraConfig* config)
|
||||
void aurora_shutdown() { aurora::shutdown(); }
|
||||
const AuroraEvent* aurora_update() { return aurora::update(); }
|
||||
bool aurora_begin_frame() { return aurora::begin_frame(); }
|
||||
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN); }
|
||||
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag); }
|
||||
void aurora_end_frame() { aurora::end_frame(AURORA_STEREO_CONTENT_TAG_UNKNOWN, {}); }
|
||||
void aurora_end_frame_tagged(uint64_t contentTag) { aurora::end_frame(contentTag, {}); }
|
||||
void aurora_end_frame_ex(uint64_t contentTag, void* imguiFrame) {
|
||||
aurora::imgui::HostFramePtr frame;
|
||||
if (imguiFrame != nullptr) {
|
||||
auto* holder = static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
|
||||
frame = std::move(*holder);
|
||||
delete holder;
|
||||
}
|
||||
aurora::end_frame(contentTag, std::move(frame));
|
||||
}
|
||||
void aurora_set_host_event_pump(bool hostPumps) {
|
||||
aurora::g_hostEventPump.store(hostPumps, std::memory_order_release);
|
||||
}
|
||||
void aurora_set_frame_log_callback(AuroraFrameLogCallback callback) {
|
||||
aurora::g_frameLogCallback.store(callback, std::memory_order_release);
|
||||
}
|
||||
extern "C" void aurora_imgui_host_frame_begin(void) {
|
||||
#ifdef AURORA_ENABLE_GX
|
||||
// ImGui's WebGPU backend creates its device objects lazily from new_frame.
|
||||
std::lock_guard gpuLock(aurora::g_rendererGpuMutex);
|
||||
#endif
|
||||
aurora::imgui::host_frame_begin(aurora::window::get_window_size());
|
||||
}
|
||||
extern "C" void* aurora_imgui_host_frame_end(void) { return new aurora::imgui::HostFramePtr(aurora::imgui::host_frame_end()); }
|
||||
extern "C" void aurora_imgui_host_frame_release(void* imguiFrame) {
|
||||
delete static_cast<aurora::imgui::HostFramePtr*>(imguiFrame);
|
||||
}
|
||||
void aurora_set_stereo_scene_anchor(const float anchorFromScene[12]) {
|
||||
aurora::set_stereo_scene_anchor(anchorFromScene);
|
||||
}
|
||||
|
||||
@@ -28,7 +28,11 @@ static std::string g_imguiLog{};
|
||||
static bool g_useSdlRenderer = false;
|
||||
// Set once ImGui::Render() has produced this frame's draw data. Interpolation encodes up to four
|
||||
// ImGui passes per frame, and every one of them used to rebuild the draw lists from scratch.
|
||||
static bool g_frameDataBuilt = false;
|
||||
static bool g_frameDataBuilt = true;
|
||||
// Host-owned frames (see imgui.hpp). Once the host begins one, aurora never calls new_frame() or
|
||||
// ImGui::Render() itself; the sealed frame carries the host's copy of the draw data instead.
|
||||
static bool g_hostFrames = false;
|
||||
static bool g_hostFrameOpen = false;
|
||||
|
||||
static std::vector<SDL_Texture*> g_sdlTextures;
|
||||
static std::vector<wgpu::Texture> g_wgpuTextures;
|
||||
@@ -179,7 +183,7 @@ void new_frame(const AuroraWindowSize& size) noexcept {
|
||||
|
||||
void render_frame_data() noexcept {
|
||||
ZoneScoped;
|
||||
if (g_frameDataBuilt) {
|
||||
if (g_frameDataBuilt || g_hostFrames) {
|
||||
return;
|
||||
}
|
||||
ImGui::Render();
|
||||
@@ -190,6 +194,11 @@ void render_frame_data() noexcept {
|
||||
|
||||
void render(const wgpu::RenderPassEncoder& pass) noexcept {
|
||||
ZoneScoped;
|
||||
if (g_hostFrames) {
|
||||
// The shared context's draw data belongs to the host's current frame now;
|
||||
// a sealed frame without a host copy has nothing safe to draw.
|
||||
return;
|
||||
}
|
||||
render_frame_data();
|
||||
|
||||
auto* data = ImGui::GetDrawData();
|
||||
@@ -205,6 +214,71 @@ void render(const wgpu::RenderPassEncoder& pass) noexcept {
|
||||
}
|
||||
}
|
||||
|
||||
struct HostFrame {
|
||||
ImDrawData data{};
|
||||
std::vector<ImDrawList*> lists;
|
||||
~HostFrame() {
|
||||
for (ImDrawList* list : lists) {
|
||||
IM_DELETE(list);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
void host_frame_begin(const AuroraWindowSize& size) noexcept {
|
||||
g_hostFrames = true;
|
||||
if (g_hostFrameOpen) {
|
||||
return;
|
||||
}
|
||||
if (!g_frameDataBuilt) {
|
||||
// aurora started this frame itself before the host took over: adopt it.
|
||||
g_hostFrameOpen = true;
|
||||
return;
|
||||
}
|
||||
new_frame(size);
|
||||
g_hostFrameOpen = true;
|
||||
}
|
||||
|
||||
HostFramePtr host_frame_end() noexcept {
|
||||
ZoneScoped;
|
||||
if (!g_hostFrameOpen) {
|
||||
host_frame_begin(window::get_window_size());
|
||||
}
|
||||
ImGui::Render();
|
||||
ImDrawData* source = ImGui::GetDrawData();
|
||||
source->FramebufferScale = ImGui::GetIO().DisplayFramebufferScale;
|
||||
auto frame = std::make_shared<HostFrame>();
|
||||
frame->data = *source;
|
||||
frame->data.CmdLists.clear();
|
||||
frame->lists.reserve(static_cast<size_t>(source->CmdListsCount));
|
||||
for (int i = 0; i < source->CmdListsCount; ++i) {
|
||||
const ImDrawList* src = source->CmdLists[i];
|
||||
ImDrawList* copy = IM_NEW(ImDrawList)(src->_Data);
|
||||
copy->CmdBuffer = src->CmdBuffer;
|
||||
copy->IdxBuffer = src->IdxBuffer;
|
||||
copy->VtxBuffer = src->VtxBuffer;
|
||||
copy->Flags = src->Flags;
|
||||
frame->lists.push_back(copy);
|
||||
frame->data.CmdLists.push_back(copy);
|
||||
}
|
||||
g_hostFrameOpen = false;
|
||||
g_frameDataBuilt = true;
|
||||
return frame;
|
||||
}
|
||||
|
||||
bool host_frames_active() noexcept { return g_hostFrames; }
|
||||
|
||||
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept { return &frame.data; }
|
||||
|
||||
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept {
|
||||
ZoneScoped;
|
||||
if (g_useSdlRenderer || data == nullptr) {
|
||||
return;
|
||||
}
|
||||
pass.PushDebugGroup("Aurora: Dear Imgui");
|
||||
ImGui_ImplWGPU_RenderDrawData(const_cast<ImDrawData*>(data), pass.Get());
|
||||
pass.PopDebugGroup();
|
||||
}
|
||||
|
||||
StereoOverlay latch_stereo_overlay() noexcept {
|
||||
std::lock_guard lock(g_stereoOverlayMutex);
|
||||
return g_stereoOverlay;
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <aurora/event.h>
|
||||
#include <memory>
|
||||
|
||||
union SDL_Event;
|
||||
struct ImDrawData;
|
||||
@@ -32,4 +33,17 @@ StereoOverlay latch_stereo_overlay() noexcept;
|
||||
// uniform for every pass, so a pass whose display size differs from the desktop's must be submitted
|
||||
// before the next pass is recorded.
|
||||
bool render_draw_data(const wgpu::RenderPassEncoder& pass, ImDrawData* data) noexcept;
|
||||
|
||||
// Host-owned ImGui frames. The host starts each frame on its own thread with host_frame_begin() and
|
||||
// closes it with host_frame_end(), which renders the frame and copies its draw data out of the shared
|
||||
// context. The copy is what the sealed frame replays, so the host may start the next frame while the
|
||||
// worker still encodes this one, and aurora stops starting frames itself once the host has begun one.
|
||||
struct HostFrame;
|
||||
using HostFramePtr = std::shared_ptr<HostFrame>;
|
||||
void host_frame_begin(const AuroraWindowSize& size) noexcept;
|
||||
HostFramePtr host_frame_end() noexcept;
|
||||
bool host_frames_active() noexcept;
|
||||
const ImDrawData* host_frame_draw_data(const HostFrame& frame) noexcept;
|
||||
// Renders a host frame's copied draw data in place of the shared context's.
|
||||
void render(const wgpu::RenderPassEncoder& pass, const ImDrawData* data) noexcept;
|
||||
} // namespace aurora::imgui
|
||||
+37
-1
@@ -588,7 +588,13 @@ the app:
|
||||
| `debug.wiicompiled.vtxpad 0` | Turns the stride padding off, to re-check a driver update |
|
||||
| `debug.wiicompiled.validation 1` | Keeps WebGPU validation and robustness on in release builds |
|
||||
| `debug.wiicompiled.inject <n>:<button>` | Presses `a`, `b`, `x`, `y`, `start`, `up`, `down`, `left` or `right` for 12 XR frames each time `<n>` changes. As a Wii Remote, `x`/`y`/`start` are 1/2/+, the directions push the Nunchuk stick, and `home`, `c` and `z` also exist. `panel` presses the settings panel's button (left Y, or both thumbsticks as a gamepad), opening or closing it (see `OPENXR.md`) |
|
||||
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
|
||||
| `debug.wiicompiled.fpslog 1` | Logs the game's rendered frame rate every 5 s, with per-frame averages of the producer's waits for the frame worker's DONE and SEALED phases and of the worker's seal, permit wait, prepare and encode stretches. A third line reports the GX thread's command ring (records, waits, busy share). A second line gives the GPU time per frame from timestamp queries on every pass (`mono` native render, `eyeL`/`eyeR` replays, `screen`, `panel`, `efbcopy`, `palette`, `peek`, plus `passes-span` from the first pass begin to the last pass end and `between-passes` for copies and idle gaps). The compositor's `VrApi` log line gives headset FPS, `GPU%`, `CPU%`, clock levels and app GPU time (`App=`) |
|
||||
|
||||
A `Config.toml` written with `adb push` (or `sed -i` in `adb shell`) belongs
|
||||
to the shell user afterwards, and the app then fails every save with EACCES
|
||||
(the launcher logs `GameStorage.prepare ... open failed`). `chmod 664` on the
|
||||
pushed file gives the app's group write access back; a file the app created
|
||||
itself never has the problem.
|
||||
|
||||
The injector makes headset tests possible with nobody wearing the headset.
|
||||
Keep the display awake, drive the menus, then take a compositor screenshot:
|
||||
@@ -687,6 +693,36 @@ plus mod code, 9% GX HLE, 6% FIFO decode, 4% memory copies, 3.5% dispatch and
|
||||
the rest. What remains on such tracks is the game's own code plus the mod's,
|
||||
which no host change shrinks; a GX thread could move about 20% of it.
|
||||
|
||||
That GX thread exists now (`runtime/include/gx_thread.h`, `[video] gx_thread`,
|
||||
on by default on Android and opt-in elsewhere). Every GX HLE override is split
|
||||
into a game-thread front, which keeps the guest-visible side effects (GXData
|
||||
shadow registers, the getters, display-list recording, the texture meta table),
|
||||
and a `_gx` back holding the aurora work and the parser state, posted through
|
||||
one ordered 16 MiB command ring; immediate-mode gather-pipe bytes travel as
|
||||
8 KiB chunks in call order. The hazard rule follows the hardware: whatever the
|
||||
SDK copied into the FIFO at call time (matrices, projection, colours, light
|
||||
objects, copy filters, layout quads, texture object registers) is snapshotted
|
||||
when posted, and whatever the GP read from memory when it reached the command
|
||||
(display lists, vertex arrays, indexed matrices, texture data) is read when the
|
||||
GX thread executes it, so `GXDrawDone` drains the ring and the frame's
|
||||
schedule, first-person anchor and policy tag are latched into the present
|
||||
record on the game thread. The desktop overlay became a game-thread-owned
|
||||
ImGui frame whose draw data Aurora copies per sealed frame, which also removed
|
||||
the frame-worker join `GXCopyDisp` used to make. With `fpslog` on, a third
|
||||
line reports the ring: records and bytes per frame, the game thread's waits
|
||||
for ring space and in drains, the GX thread's busy share and any exceptions
|
||||
it caught. A texture or matrix that is wrong only with the thread on is a
|
||||
hazard-rule violation (a front reading guest memory the game rewrites before
|
||||
the GX thread runs, or a back writing guest memory). Measured on the same
|
||||
automated Grand Prix start at `render_scale` 0.75, same build, switched by the
|
||||
config key: with the thread off the crowded first half minute ran at 52 to
|
||||
56 fps before settling at 60; with it on the same stretch ran at 56.5 in the
|
||||
window that includes the countdown and 60.0 in every window after, while the
|
||||
ring carried 4.5k to 6.2k records (250 to 380 KiB) per frame, the game thread
|
||||
waited under 0.1 ms per frame in its two `GXDrawDone` drains and never for
|
||||
ring space, and the GX thread was 25 to 40% busy. Retro Rewind's menus were
|
||||
unaffected (prewarm 5.2 s, 60 fps).
|
||||
|
||||
Verified on device since: the menus on the virtual screen, controller input
|
||||
(the user has driven races), and an immersive Grand Prix start with all 12
|
||||
racers rendering correctly. Not yet verified: stereo comfort and scale,
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include "settings_overlay.h"
|
||||
#include "runtime_config.h"
|
||||
#include "fiber_manager.h"
|
||||
#include "gx_thread.h"
|
||||
|
||||
#include <aurora/aurora.h>
|
||||
#include <aurora/event.h>
|
||||
@@ -142,7 +143,11 @@ inline bool BeginAuroraFrame() {
|
||||
if (!aurora_begin_frame()) {
|
||||
return false;
|
||||
}
|
||||
ApplyPendingMkwDynamicAspectSurface();
|
||||
// The viewport policy writes guest memory (EGG screen records), so the GX
|
||||
// thread leaves it to the game thread's present path.
|
||||
if (!GxThread::IsGxThread()) {
|
||||
ApplyPendingMkwDynamicAspectSurface();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
#pragma once
|
||||
// GX thread: the host side of the GX pipeline (state tracking, FIFO parsing,
|
||||
// display-list scanning, texture object resolution and every aurora GX call)
|
||||
// runs on its own thread, fed by an ordered command ring the game thread posts
|
||||
// to. Each GX HLE override is split into a game-thread front (guest-visible
|
||||
// side effects: shadow registers, getters, display-list recording) and a
|
||||
// GX-thread back (the aurora work). When the thread is disabled, Post() runs
|
||||
// the back inline, so the split is behaviour-preserving in both modes.
|
||||
//
|
||||
// Data hazards follow the hardware: anything the GX library copies into the
|
||||
// FIFO at call time (immediate-mode vertices, matrices, colours, light objects,
|
||||
// texture object registers) is snapshotted at post time; anything the GP reads
|
||||
// from memory when it reaches the command (display lists, vertex arrays,
|
||||
// indexed matrices, texture data) is read when the GX thread executes it.
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
|
||||
namespace GxThread {
|
||||
|
||||
// Decides whether posted work runs on the GX thread. Read once, before GXInit.
|
||||
void Configure(bool enabled);
|
||||
bool Enabled() noexcept;
|
||||
// True on the consumer thread.
|
||||
bool IsGxThread() noexcept;
|
||||
void Start();
|
||||
// Executes everything posted so far and joins the thread.
|
||||
void Stop();
|
||||
// Game thread: blocks until every posted record has executed (GXDrawDone).
|
||||
void Drain();
|
||||
// Publishes the open immediate-mode FIFO chunk. Post() does this itself.
|
||||
void FlushFifo();
|
||||
// Invoked at bounded intervals while the game thread blocks in Drain() or on
|
||||
// a full ring, so guest timing (VI retraces, OS alarms) keeps running.
|
||||
void SetWaitCallback(void (*callback)());
|
||||
// Native thread id of the consumer (Android: gettid), 0 until started.
|
||||
uint32_t NativeThreadId() noexcept;
|
||||
// One line for the frame-rate log; resets the window counters.
|
||||
std::string FormatStatsAndReset(double windowSeconds, uint32_t frames);
|
||||
|
||||
// Immediate-mode write-gather bytes. Only valid when Enabled().
|
||||
void PostFifoWord(uint32_t value, uint32_t sizeBytes);
|
||||
void PostFifoBytes(const uint8_t* data, uint32_t sizeBytes);
|
||||
|
||||
namespace detail {
|
||||
using Invoke = void (*)(const uint8_t* payload, uint32_t payloadBytes);
|
||||
void PostRecord(Invoke invoke, const void* payload, uint32_t payloadBytes);
|
||||
|
||||
template <typename... Ts>
|
||||
struct Pack;
|
||||
template <>
|
||||
struct Pack<> {
|
||||
template <typename F, typename... Prev>
|
||||
void Call(F f, Prev... prev) const {
|
||||
f(prev...);
|
||||
}
|
||||
};
|
||||
template <typename T, typename... Ts>
|
||||
struct Pack<T, Ts...> {
|
||||
T head;
|
||||
Pack<Ts...> tail;
|
||||
template <typename F, typename... Prev>
|
||||
void Call(F f, Prev... prev) const {
|
||||
tail.Call(f, prev..., head);
|
||||
}
|
||||
};
|
||||
template <typename... Ts>
|
||||
struct BuildPack;
|
||||
template <>
|
||||
struct BuildPack<> {
|
||||
static Pack<> Make() { return {}; }
|
||||
};
|
||||
template <typename T, typename... Ts>
|
||||
struct BuildPack<T, Ts...> {
|
||||
static Pack<T, Ts...> Make(T head, Ts... tail) {
|
||||
return Pack<T, Ts...>{head, BuildPack<Ts...>::Make(tail...)};
|
||||
}
|
||||
};
|
||||
template <typename R, typename... Params>
|
||||
struct CallRecord {
|
||||
R (*fn)(Params...);
|
||||
Pack<Params...> args;
|
||||
};
|
||||
template <typename R, typename... Params>
|
||||
void InvokeCall(const uint8_t* payload, uint32_t) {
|
||||
CallRecord<R, Params...> record;
|
||||
std::memcpy(&record, payload, sizeof(record));
|
||||
record.args.Call(record.fn);
|
||||
}
|
||||
} // namespace detail
|
||||
|
||||
// Posts fn(args...) to the GX thread, or runs it now when the thread is off.
|
||||
// Parameters must be trivially copyable values (no pointers into guest memory
|
||||
// that the game may rewrite before the GX thread reads them).
|
||||
template <typename R, typename... Params, typename... Args>
|
||||
inline void Post(R (*fn)(Params...), Args&&... args) {
|
||||
if (!Enabled()) {
|
||||
fn(static_cast<Params>(args)...);
|
||||
return;
|
||||
}
|
||||
detail::CallRecord<R, Params...> record{fn, detail::BuildPack<Params...>::Make(static_cast<Params>(args)...)};
|
||||
detail::PostRecord(&detail::InvokeCall<R, Params...>, &record, sizeof(record));
|
||||
}
|
||||
|
||||
} // namespace GxThread
|
||||
@@ -47,6 +47,7 @@ struct RuntimeUserConfig {
|
||||
std::optional<bool> textureReplacements;
|
||||
std::optional<bool> textureDumps;
|
||||
std::optional<bool> showFps;
|
||||
std::optional<bool> gxThread;
|
||||
std::optional<uint32_t> disabledPostProcessingPaths;
|
||||
std::optional<bool> vrEnabled;
|
||||
std::optional<bool> vrRequired;
|
||||
@@ -410,6 +411,10 @@ inline void EnsureConfigFile() {
|
||||
"skip_unready_pipelines = true\n"
|
||||
"disable_copy_filter = true\n"
|
||||
"show_fps = true\n"
|
||||
"# Run the host side of the GX pipeline (state tracking, FIFO parsing,\n"
|
||||
"# texture uploads) on its own thread. On by default on the Quest, where\n"
|
||||
"# the game thread is the bottleneck; opt-in elsewhere.\n"
|
||||
"# gx_thread = true\n"
|
||||
"# Dolphin-style custom textures. When enabled, the renderer indexes\n"
|
||||
"# texture_replacements/ next to this file at startup and substitutes\n"
|
||||
"# any tex1_<W>x<H>_<hash>[_<tlut hash>]_<format>.dds or .png it finds\n"
|
||||
@@ -641,6 +646,7 @@ inline RuntimeUserConfig ParseConfigDocument(const toml::value& document) {
|
||||
config.skipUnreadyPipelines = FindConfigValue<bool>(document, "video", "skip_unready_pipelines");
|
||||
config.disableCopyFilter = FindConfigValue<bool>(document, "video", "disable_copy_filter");
|
||||
config.showFps = FindConfigValue<bool>(document, "video", "show_fps");
|
||||
config.gxThread = FindConfigValue<bool>(document, "video", "gx_thread");
|
||||
config.textureReplacements = FindConfigValue<bool>(document, "video", "texture_replacements");
|
||||
config.textureDumps = FindConfigValue<bool>(document, "video", "texture_dumps");
|
||||
if (auto value = FindConfigUint(document, "video", "disabled_post_processing_paths");
|
||||
@@ -1294,6 +1300,16 @@ inline bool ShowFps(bool fallback = true) {
|
||||
return Get().showFps.value_or(fallback);
|
||||
}
|
||||
|
||||
// Runs the host side of the GX pipeline on its own thread (gx_thread.h). On by
|
||||
// default on the Quest, where the game thread is the bottleneck; opt-in elsewhere.
|
||||
inline bool GxThread() {
|
||||
#if defined(__ANDROID__)
|
||||
return Get().gxThread.value_or(true);
|
||||
#else
|
||||
return Get().gxThread.value_or(false);
|
||||
#endif
|
||||
}
|
||||
|
||||
inline bool TextureReplacements(bool fallback = false) {
|
||||
return Get().textureReplacements.value_or(fallback);
|
||||
}
|
||||
|
||||
@@ -62,69 +62,84 @@ void InvalidateEfbCopyDestinationsForRange(uint32_t addr, uint32_t size) {
|
||||
// Preserve FIFO ordering: the destroy command is emitted before any later
|
||||
// texture load that can consume the freshly flushed RAM bytes.
|
||||
for (const uint32_t copyAddr : retired) {
|
||||
GXDestroyCopyTex(GuestToHostPtr(copyAddr));
|
||||
GxThread::Post(&GxHostDestroyCopyTex_gx, copyAddr);
|
||||
}
|
||||
}
|
||||
|
||||
void GxHostDestroyCopyTex_gx(uint32_t copyAddr) { GXDestroyCopyTex(GuestToHostPtr(copyAddr)); }
|
||||
|
||||
// ============================================================================
|
||||
// Display Copy Source/Destination
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetDispCopySrc_8016f438(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
|
||||
static void GX__SetDispCopySrc_8016f438_gx(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
|
||||
GXSetDispCopySrc((u16)l, (u16)t, (u16)w, (u16)h);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f438, GX__SetDispCopySrc_8016f438, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f438, GX__SetDispCopySrc_8016f438, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
|
||||
|
||||
extern "C" void GX__SetDispCopyDst_8016f4b8(uint32_t w, uint32_t h) { GXSetDispCopyDst((u16)w, (u16)h); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f4b8, GX__SetDispCopyDst_8016f4b8, (uint32_t w, uint32_t h), (w, h));
|
||||
static void GX__SetDispCopyDst_8016f4b8_gx(uint32_t w, uint32_t h) { GXSetDispCopyDst((u16)w, (u16)h); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f4b8, GX__SetDispCopyDst_8016f4b8, (uint32_t w, uint32_t h), (w, h));
|
||||
|
||||
// ============================================================================
|
||||
// Texture Copy Source/Destination
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetTexCopySrc_8016f478(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
|
||||
static void GX__SetTexCopySrc_gx(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
|
||||
GXSetTexCopySrc((u16)l, (u16)t, (u16)w, (u16)h);
|
||||
}
|
||||
extern "C" void GX__SetTexCopySrc_8016f478(uint32_t l, uint32_t t, uint32_t w, uint32_t h) {
|
||||
GxThread::Post(&GX__SetTexCopySrc_gx, l, t, w, h);
|
||||
g_texCopyState.srcLeft=(u16)l; g_texCopyState.srcTop=(u16)t;
|
||||
g_texCopyState.srcWidth=(u16)w; g_texCopyState.srcHeight=(u16)h;
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f478, GX__SetTexCopySrc_8016f478, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
|
||||
|
||||
extern "C" void GX__SetTexCopyDst_8016f4dc(uint32_t w, uint32_t h, uint32_t f, uint32_t m) {
|
||||
static void GX__SetTexCopyDst_gx(uint32_t w, uint32_t h, uint32_t f, uint32_t m) {
|
||||
GXSetTexCopyDst((u16)w, (u16)h, (GXTexFmt)f, (GXBool)m);
|
||||
}
|
||||
extern "C" void GX__SetTexCopyDst_8016f4dc(uint32_t w, uint32_t h, uint32_t f, uint32_t m) {
|
||||
GxThread::Post(&GX__SetTexCopyDst_gx, w, h, f, m);
|
||||
g_texCopyState.dstWidth=(u16)w; g_texCopyState.dstHeight=(u16)h;
|
||||
g_texCopyState.dstFormat=f; g_texCopyState.dstMipmap=m;
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f4dc, GX__SetTexCopyDst_8016f4dc, (uint32_t w, uint32_t h, uint32_t f, uint32_t m), (w, h, f, m));
|
||||
|
||||
struct GxCopyFilterSnapshot {
|
||||
uint8_t sp[12][2];
|
||||
uint8_t vfb[7];
|
||||
};
|
||||
static void GX__SetCopyFilter_gx(uint32_t aa, uint32_t vf, GxCopyFilterSnapshot filter) {
|
||||
GXSetCopyFilter((GXBool)aa, filter.sp, (GXBool)vf, filter.vfb);
|
||||
}
|
||||
extern "C" void GX__SetCopyFilter_8016fa40(uint32_t aa, uint32_t spa, uint32_t vf, uint32_t vfa) {
|
||||
uint8_t sp[12][2]={}, vfb[7]={};
|
||||
if(spa) std::memcpy(sp, GuestToHostPtr(spa, 24), 24);
|
||||
if(vfa) std::memcpy(vfb, GuestToHostPtr(vfa, 7), 7);
|
||||
GXSetCopyFilter((GXBool)aa, sp, (GXBool)vf, vfb);
|
||||
GxCopyFilterSnapshot filter{};
|
||||
if(spa) std::memcpy(filter.sp, GuestToHostPtr(spa, 24), 24);
|
||||
if(vfa) std::memcpy(filter.vfb, GuestToHostPtr(vfa, 7), 7);
|
||||
GxThread::Post(&GX__SetCopyFilter_gx, aa, vf, filter);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016fa40, GX__SetCopyFilter_8016fa40, (uint32_t aa, uint32_t spa, uint32_t vf, uint32_t vfa), (aa, spa, vf, vfa));
|
||||
|
||||
extern "C" void GX__SetDispCopyGamma_8016fc24(uint32_t g) { GXSetDispCopyGamma((GXGamma)g); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016fc24, GX__SetDispCopyGamma_8016fc24, (uint32_t g), (g));
|
||||
static void GX__SetDispCopyGamma_8016fc24_gx(uint32_t g) { GXSetDispCopyGamma((GXGamma)g); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016fc24, GX__SetDispCopyGamma_8016fc24, (uint32_t g), (g));
|
||||
|
||||
// ============================================================================
|
||||
// Copy Execution
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__CopyDisp_8016fc38(uint32_t da, uint32_t c) {
|
||||
static void GX__CopyDisp_gx(uint32_t da, uint32_t c) {
|
||||
EnsureAuroraFrameActive();
|
||||
// GX copies are FIFO-ordered on hardware. Drain submitted draws before
|
||||
// resolving the EFB so high-level copies see the same contents.
|
||||
GXDrawDone();
|
||||
GXCopyDisp(GuestToHostPtr(da), (GXBool)c);
|
||||
// No second GXDrawDone here: the frame-worker wait below is for the DONE
|
||||
// phase, which strictly subsumes the drain this call would perform.
|
||||
}
|
||||
extern "C" void GX__CopyDisp_8016fc38(uint32_t da, uint32_t c) {
|
||||
GxThread::Post(&GX__CopyDisp_gx, da, c);
|
||||
++g_gxFrameCount;
|
||||
VI_HLE_SetXfbReady(da);
|
||||
// Present immediately so post-copy draws don't leak into this frame. Join at the DONE phase
|
||||
// (not the cheaper SEALED phase GXDrawDone waits for) because ImGui's draw lists, owned by
|
||||
// Aurora's render worker, replay during encode; aurora_end_frame would join here anyway.
|
||||
aurora_wait_for_frame_worker();
|
||||
// Present immediately so post-copy draws don't leak into this frame. The
|
||||
// overlay draws into the game thread's own ImGui frame, whose draw data
|
||||
// the seal copies, so no frame-worker join is needed here.
|
||||
settings_overlay::Draw();
|
||||
// Seal, pace to the VI retrace boundary (Aurora renders the sealed frame
|
||||
// during the wait), and pre-warm the next frame.
|
||||
@@ -134,14 +149,15 @@ extern "C" void GX__CopyDisp_8016fc38(uint32_t da, uint32_t c) {
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016fc38, GX__CopyDisp_8016fc38, (uint32_t da, uint32_t c), (da, c));
|
||||
|
||||
|
||||
extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) {
|
||||
static void GX__CopyTex_gx(uint32_t da, uint32_t c, uint32_t srcLeft, uint32_t srcTop, uint32_t srcWidth,
|
||||
uint32_t srcHeight) {
|
||||
EnsureAuroraFrameActive();
|
||||
// Match GX FIFO ordering: texture copies observe all prior draws.
|
||||
GXDrawDone();
|
||||
const uint16_t rawSrcLeft = g_texCopyState.srcLeft;
|
||||
const uint16_t rawSrcTop = g_texCopyState.srcTop;
|
||||
const uint16_t rawSrcWidth = g_texCopyState.srcWidth;
|
||||
const uint16_t rawSrcHeight = g_texCopyState.srcHeight;
|
||||
const uint16_t rawSrcLeft = (uint16_t)srcLeft;
|
||||
const uint16_t rawSrcTop = (uint16_t)srcTop;
|
||||
const uint16_t rawSrcWidth = (uint16_t)srcWidth;
|
||||
const uint16_t rawSrcHeight = (uint16_t)srcHeight;
|
||||
|
||||
// Keep the source in guest EFB coordinates. Aurora maps it to the scaled
|
||||
// EFB exactly once, matching Dolphin's ConvertEFBRectangle path.
|
||||
@@ -153,9 +169,13 @@ extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) {
|
||||
// auto-downloaded, so guest reads see stale RAM; call aurora_flush_efb_copies_to_ram if a
|
||||
// copy needs reading back.
|
||||
GXCopyTex(GuestToHostPtr(da), (GXBool)c);
|
||||
GXSetTexCopySrc(rawSrcLeft, rawSrcTop, rawSrcWidth, rawSrcHeight);
|
||||
}
|
||||
extern "C" void GX__CopyTex_8016fd74(uint32_t da, uint32_t c) {
|
||||
GxThread::Post(&GX__CopyTex_gx, da, c, g_texCopyState.srcLeft, g_texCopyState.srcTop,
|
||||
g_texCopyState.srcWidth, g_texCopyState.srcHeight);
|
||||
RememberEfbCopyDestination(
|
||||
da, GXGetTexBufferSize(g_texCopyState.dstWidth, g_texCopyState.dstHeight,
|
||||
g_texCopyState.dstFormat, GX_FALSE, 0));
|
||||
GXSetTexCopySrc(rawSrcLeft, rawSrcTop, rawSrcWidth, rawSrcHeight);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016fd74, GX__CopyTex_8016fd74, (uint32_t da, uint32_t c), (da, c));
|
||||
@@ -128,7 +128,7 @@ static inline void AppendLytQuadVertices(uint8_t* packet, uint32_t& pos, float x
|
||||
AppendLytQuadVertex(packet, pos, x0, y1, texCoordAddr, texCoordCount, 16, colors, 2);
|
||||
}
|
||||
|
||||
static bool CanSubmitLytDrawDirect(int texCoordCount, const uint32_t* colors) {
|
||||
static bool CanSubmitLytDrawDirect(int texCoordCount, bool hasColors) {
|
||||
if (texCoordCount < 0 || texCoordCount > 8) {
|
||||
return false;
|
||||
}
|
||||
@@ -139,7 +139,6 @@ static bool CanSubmitLytDrawDirect(int texCoordCount, const uint32_t* colors) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const bool hasColors = colors != nullptr;
|
||||
const auto& clrFmt = g_hleGxState.vtxAttrFmt[GX_VTXFMT0][GX_VA_CLR0];
|
||||
if (hasColors) {
|
||||
if (g_hleGxState.vtxDesc[GX_VA_CLR0] != GX_DIRECT ||
|
||||
@@ -179,33 +178,30 @@ static bool CanSubmitLytDrawDirect(int texCoordCount, const uint32_t* colors) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool SubmitLytDrawDirect(float x0, float y0, float x1, float y1, int texCoordCount,
|
||||
uint32_t texCoordAddr, const uint32_t* colors) {
|
||||
if (!CanSubmitLytDrawDirect(texCoordCount, colors)) {
|
||||
return false;
|
||||
// One nw4r::lyt quad as a GX draw packet (3-byte header, 4 vertices), built on
|
||||
// the game thread from the guest's layout data and submitted on the GX thread,
|
||||
// whose descriptor state decides between the raw-draw fast path and the packet
|
||||
// parser.
|
||||
struct GxLytQuadPacket {
|
||||
uint32_t bytes = 0;
|
||||
int32_t texCoordCount = 0;
|
||||
bool hasColors = false;
|
||||
uint8_t data[3u + 4u * (8u + 4u + 8u * 8u)];
|
||||
};
|
||||
|
||||
static void GxLytQuad_gx(GxLytQuadPacket packet) {
|
||||
if (CanSubmitLytDrawDirect(packet.texCoordCount, packet.hasColors)) {
|
||||
EnsureAuroraFrameActive();
|
||||
ApplyAuroraVtxDesc();
|
||||
ApplyAuroraVtxAttrFmtForDisplayList(GX_VTXFMT0, false);
|
||||
EnsureDefaultGxAlphaCompare();
|
||||
if (aurora::gx::fifo::submit_raw_draw(GX_QUADS, GX_VTXFMT0, packet.data + 3, 4, packet.bytes - 3u)) {
|
||||
GXMarkFrameWork();
|
||||
SyncAppliedVtxStateFromHleReal();
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
EnsureAuroraFrameActive();
|
||||
|
||||
ApplyAuroraVtxDesc();
|
||||
|
||||
ApplyAuroraVtxAttrFmtForDisplayList(GX_VTXFMT0, false);
|
||||
|
||||
EnsureDefaultGxAlphaCompare();
|
||||
|
||||
|
||||
std::array<uint8_t, 4u * (8u + 4u + 8u * 8u)> vertices{};
|
||||
uint32_t pos = 0;
|
||||
AppendLytQuadVertices(vertices.data(), pos, x0, y0, x1, y1, texCoordAddr, texCoordCount, colors);
|
||||
|
||||
if (!aurora::gx::fifo::submit_raw_draw(GX_QUADS, GX_VTXFMT0, vertices.data(), 4, pos)) {
|
||||
return false;
|
||||
}
|
||||
GXMarkFrameWork();
|
||||
|
||||
SyncAppliedVtxStateFromHleReal();
|
||||
return true;
|
||||
SubmitLytDrawPacket(packet.data, packet.bytes);
|
||||
}
|
||||
|
||||
static inline void EmitLytDrawQuad(uint32_t posAddr, uint32_t sizeAddr, int texCoordCount,
|
||||
@@ -217,10 +213,6 @@ static inline void EmitLytDrawQuad(uint32_t posAddr, uint32_t sizeAddr, int texC
|
||||
const float x1 = static_cast<float>(x0 + Memory::ReadFloat32(sizeAddr));
|
||||
const float y1 = static_cast<float>(y0 - Memory::ReadFloat32(sizeAddr + 4));
|
||||
|
||||
if (SubmitLytDrawDirect(x0, y0, x1, y1, texCoordCount, texCoordAddr, colors)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// GX has exactly 8 texture coordinates, so nw4r::lyt cannot ask for more.
|
||||
// The fixed packet buffer below is sized for that maximum; bail rather than
|
||||
// overrun it if the guest ever hands us something else.
|
||||
@@ -228,12 +220,15 @@ static inline void EmitLytDrawQuad(uint32_t posAddr, uint32_t sizeAddr, int texC
|
||||
return;
|
||||
}
|
||||
|
||||
std::array<uint8_t, 3u + 4u * (8u + 4u + 8u * 8u)> packet{};
|
||||
GxLytQuadPacket packet{};
|
||||
packet.texCoordCount = texCoordCount;
|
||||
packet.hasColors = colors != nullptr;
|
||||
uint32_t pos = 0;
|
||||
packet[pos++] = GX_DRAW_QUADS_CMD | GX_VTXFMT0;
|
||||
BigEndian::Append16(packet.data(), pos, 4);
|
||||
AppendLytQuadVertices(packet.data(), pos, x0, y0, x1, y1, texCoordAddr, texCoordCount, colors);
|
||||
SubmitLytDrawPacket(packet.data(), pos);
|
||||
packet.data[pos++] = GX_DRAW_QUADS_CMD | GX_VTXFMT0;
|
||||
BigEndian::Append16(packet.data, pos, 4);
|
||||
AppendLytQuadVertices(packet.data, pos, x0, y0, x1, y1, texCoordAddr, texCoordCount, colors);
|
||||
packet.bytes = pos;
|
||||
GxThread::Post(&GxLytQuad_gx, packet);
|
||||
}
|
||||
|
||||
static uint32_t SubmitDLVertex(const uint8_t* ptr, GXVtxFmt vtxfmt, const GXAttrType* sourceVtxDesc) {
|
||||
@@ -1293,7 +1288,8 @@ extern "C" void GxNotifyDisplayListMemoryWrite(uint32_t addr, uint32_t size) {
|
||||
GxGuestWrite::NotifyWrite(addr, size);
|
||||
}
|
||||
|
||||
extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes) {
|
||||
// The list itself is read when the GX thread reaches the call, as the GP does.
|
||||
void GX__CallDisplayList_gx(uint32_t listAddr, uint32_t nbytes) {
|
||||
if (nbytes == 0 || listAddr == 0) return;
|
||||
try {
|
||||
const uint8_t* list = static_cast<const uint8_t*>(GuestToHostPtr(listAddr, nbytes));
|
||||
@@ -1487,6 +1483,9 @@ extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes)
|
||||
} catch (...) {}
|
||||
}
|
||||
|
||||
extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes) {
|
||||
GxThread::Post(&GX__CallDisplayList_gx, listAddr, nbytes);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172F64, GX__CallDisplayList_80172f64, (uint32_t listAddr, uint32_t nbytes), (listAddr, nbytes));
|
||||
|
||||
extern "C" void nw4r__lyt__detail__DrawQuad_800847c0(CpuContext* ctx) {
|
||||
|
||||
@@ -79,10 +79,10 @@ extern "C" void EGG__LightTexture__SetupTevFinish_HLE_8022e2bc(CpuContext* ctx)
|
||||
if (stageCount != 0) {
|
||||
uint32_t remainder = tevCount % stageCount;
|
||||
if (remainder > 0) {
|
||||
const GXColor black{0, 0, 0, 255};
|
||||
constexpr uint32_t kBlack = 0x000000FFu;
|
||||
while (remainder < stageCount) {
|
||||
GXSetTevColor(static_cast<GXTevRegID>(remainder + 1), black);
|
||||
GXSetTevKColor(static_cast<GXTevKColorID>(remainder), black);
|
||||
GxThread::Post(&GX__SetTevColor_gx, remainder + 1, kBlack);
|
||||
GxThread::Post(&GX__SetTevKColor_gx, remainder, kBlack);
|
||||
++remainder;
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#include "gx_stream_common.h"
|
||||
#include "gx_cp_decode.h"
|
||||
#include "isa/big_endian.h"
|
||||
#include "runtime_log.h"
|
||||
|
||||
// Opcode constants and the stream helpers this file shares with gx_dl.cpp /
|
||||
// gx_vertex.cpp; see gx_stream_common.h.
|
||||
@@ -346,7 +347,9 @@ void SubmitAttribute(GXAttr attr, float* comps, const VtxAttrFmt& fmt, const u32
|
||||
// `val` is a raw big-endian bit pattern: the FIFO stream is type-agnostic, and
|
||||
// the float entry point converts before it gets here.
|
||||
void HleFifoWrite(u32 val, uint32_t sizeBytes) {
|
||||
const bool recordOnly = IsDisplayListActive();
|
||||
// Display-list recording is game-thread state; bytes only reach the GX
|
||||
// thread when nothing is being recorded, so it must not consult it.
|
||||
const bool recordOnly = !GxThread::IsGxThread() && IsDisplayListActive();
|
||||
if (recordOnly) {
|
||||
WriteDisplayListData(val, sizeBytes);
|
||||
return;
|
||||
@@ -480,7 +483,7 @@ void HleFifoWrite(u32 val, uint32_t sizeBytes) {
|
||||
const uint32_t listSize = ReadBE32(data + 5);
|
||||
if (!consumeBytes(9, sink)) break;
|
||||
if (listAddr != 0 && listSize > 0) {
|
||||
GX__CallDisplayList_80172f64(listAddr, listSize);
|
||||
GX__CallDisplayList_gx(listAddr, listSize);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
@@ -682,7 +685,8 @@ static uint32_t ApplyFifoPacketsDirect(const uint8_t* data, uint32_t sizeBytes)
|
||||
while (offset < sizeBytes) {
|
||||
// Re-tested per packet, not once per burst: nothing currently re-enters GX HLE mid-walk,
|
||||
// but if it ever does, breaking here just hands the remainder to the ring.
|
||||
if (IsDisplayListActive() || g_hleGxState.inBegin || g_hleGxState.fifoByteCount != 0) {
|
||||
if ((!GxThread::IsGxThread() && IsDisplayListActive()) || g_hleGxState.inBegin ||
|
||||
g_hleGxState.fifoByteCount != 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -775,17 +779,50 @@ static bool WriteDisplayListBurst(const uint8_t* data, uint32_t sizeBytes) {
|
||||
return true;
|
||||
}
|
||||
|
||||
extern "C" void GX_HLE_FIFO_WriteBurst(const uint8_t* data, uint32_t sizeBytes) {
|
||||
if (data == nullptr || sizeBytes == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (IsDisplayListActive() && WriteDisplayListBurst(data, sizeBytes)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// GX-thread side of the write-gather pipe (or inline when the thread is off).
|
||||
extern "C" void GxFifoConsumeBytes(const uint8_t* data, uint32_t sizeBytes) {
|
||||
const uint32_t applied = ApplyFifoPacketsDirect(data, sizeBytes);
|
||||
if (applied < sizeBytes) {
|
||||
HleFifoWriteBurstChunked(data + applied, sizeBytes - applied);
|
||||
}
|
||||
}
|
||||
|
||||
// Game-thread fronts of the write-gather pipe. Display-list recording is
|
||||
// resolved here (it writes guest memory); everything else is parsed now or
|
||||
// posted to the GX thread as raw bytes, in call order.
|
||||
static inline void GxFifoFrontWrite(u32 val, uint32_t sizeBytes) {
|
||||
if (IsDisplayListActive()) {
|
||||
WriteDisplayListData(val, sizeBytes);
|
||||
return;
|
||||
}
|
||||
if (GxThread::Enabled()) {
|
||||
GxThread::PostFifoWord(val, sizeBytes);
|
||||
return;
|
||||
}
|
||||
HleFifoWrite(val, sizeBytes);
|
||||
}
|
||||
|
||||
extern "C" void GX_HLE_FIFO_WriteFloat(float val) {
|
||||
u32 raw; std::memcpy(&raw, &val, 4);
|
||||
try { GxFifoFrontWrite(raw, 4); } catch (...) { RT_LOGF(RT_TAG_GX, "FIFO write float failed\n"); }
|
||||
}
|
||||
extern "C" void GX_HLE_FIFO_Write32(uint32_t val) { GxFifoFrontWrite(val, 4); }
|
||||
extern "C" void GX_HLE_FIFO_Write16(uint16_t val) { GxFifoFrontWrite(static_cast<u32>(val), 2); }
|
||||
extern "C" void GX_HLE_FIFO_Write8(uint8_t val) { GxFifoFrontWrite(static_cast<u32>(val), 1); }
|
||||
|
||||
extern "C" void GX_HLE_FIFO_WriteBurst(const uint8_t* data, uint32_t sizeBytes) {
|
||||
if (data == nullptr || sizeBytes == 0) {
|
||||
return;
|
||||
}
|
||||
if (IsDisplayListActive()) {
|
||||
if (!WriteDisplayListBurst(data, sizeBytes)) {
|
||||
HleFifoWriteBurstChunked(data, sizeBytes);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (GxThread::Enabled()) {
|
||||
GxThread::PostFifoBytes(data, sizeBytes);
|
||||
return;
|
||||
}
|
||||
GxFifoConsumeBytes(data, sizeBytes);
|
||||
}
|
||||
@@ -68,7 +68,10 @@ void BeginNextAuroraFrameWithRetry(std::chrono::milliseconds timeout) {
|
||||
const auto deadline = std::chrono::steady_clock::now() + timeout;
|
||||
uint32_t attempts = 0;
|
||||
while (std::chrono::steady_clock::now() < deadline) {
|
||||
UpdateAuroraAndProcessEvents();
|
||||
// SDL is pumped by the window's thread; the GX thread only retries the begin.
|
||||
if (!GxThread::IsGxThread()) {
|
||||
UpdateAuroraAndProcessEvents();
|
||||
}
|
||||
++attempts;
|
||||
if (BeginAuroraFrame()) {
|
||||
g_auroraFrameActive.store(true, std::memory_order_release);
|
||||
|
||||
@@ -5,22 +5,29 @@
|
||||
// Indirect Texture Stages
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetNumIndStages_80171b38(uint32_t n) { GXSetNumIndStages((u8)n); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171b38, GX__SetNumIndStages_80171b38, (uint32_t n), (n));
|
||||
static void GX__SetNumIndStages_80171b38_gx(uint32_t n) { GXSetNumIndStages((u8)n); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171b38, GX__SetNumIndStages_80171b38, (uint32_t n), (n));
|
||||
|
||||
extern "C" void GX__SetIndTexOrder_80171a6c(uint32_t s, uint32_t c, uint32_t m) {
|
||||
static void GX__SetIndTexOrder_80171a6c_gx(uint32_t s, uint32_t c, uint32_t m) {
|
||||
GXSetIndTexOrder((GXIndTexStageID)s, (GXTexCoordID)(c==0xFFu?0:c), (GXTexMapID)(m==0xFFu?0:m));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171a6c, GX__SetIndTexOrder_80171a6c, (uint32_t s, uint32_t c, uint32_t m), (s, c, m));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171a6c, GX__SetIndTexOrder_80171a6c, (uint32_t s, uint32_t c, uint32_t m), (s, c, m));
|
||||
|
||||
extern "C" void GX__SetIndTexCoordScale_80171968(uint32_t s, uint32_t ss, uint32_t ts) {
|
||||
static void GX__SetIndTexCoordScale_80171968_gx(uint32_t s, uint32_t ss, uint32_t ts) {
|
||||
GXSetIndTexCoordScale((GXIndTexStageID)s, (GXIndTexScale)ss, (GXIndTexScale)ts);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171968, GX__SetIndTexCoordScale_80171968, (uint32_t s, uint32_t ss, uint32_t ts), (s, ss, ts));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171968, GX__SetIndTexCoordScale_80171968, (uint32_t s, uint32_t ss, uint32_t ts), (s, ss, ts));
|
||||
|
||||
struct GxIndTexMtxSnapshot {
|
||||
float m[6];
|
||||
};
|
||||
static void GX__SetIndTexMtx_gx(uint32_t id, GxIndTexMtxSnapshot mtx, uint32_t se) {
|
||||
GXSetIndTexMtx((GXIndTexMtxID)id, mtx.m, (s8)se);
|
||||
}
|
||||
extern "C" void GX__SetIndTexMtx_80171814(uint32_t id, uint32_t ma, uint32_t se) {
|
||||
float m[6]; for(int i=0; i<6; ++i) m[i]=Memory::ReadFloat32(ma+i*4);
|
||||
GXSetIndTexMtx((GXIndTexMtxID)id, m, (s8)se);
|
||||
GxIndTexMtxSnapshot mtx{};
|
||||
for(int i=0; i<6; ++i) mtx.m[i]=Memory::ReadFloat32(ma+i*4);
|
||||
GxThread::Post(&GX__SetIndTexMtx_gx, id, mtx, se);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171814, GX__SetIndTexMtx_80171814, (uint32_t id, uint32_t ma, uint32_t se), (id, ma, se));
|
||||
|
||||
@@ -28,18 +35,18 @@ PPC_NATIVE_OVERRIDE_VOID(80171814, GX__SetIndTexMtx_80171814, (uint32_t id, uint
|
||||
// TEV Indirect Texture Control
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetTevDirect_80171b58(uint32_t s) { GXSetTevDirect((GXTevStageID)s); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171b58, GX__SetTevDirect_80171b58, (uint32_t s), (s));
|
||||
static void GX__SetTevDirect_80171b58_gx(uint32_t s) { GXSetTevDirect((GXTevStageID)s); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171b58, GX__SetTevDirect_80171b58, (uint32_t s), (s));
|
||||
|
||||
extern "C" void GX__SetTevIndWarp_80171ba0(uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms) {
|
||||
static void GX__SetTevIndWarp_80171ba0_gx(uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms) {
|
||||
GXSetTevIndWarp((GXTevStageID)std::min(ts, 15u), (GXIndTexStageID)std::min(is, 3u),
|
||||
(GXBool)so, (GXBool)rm, (GXIndTexMtxID)ms);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171ba0, GX__SetTevIndWarp_80171ba0, (uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms), (ts, is, so, rm, ms));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171ba0, GX__SetTevIndWarp_80171ba0, (uint32_t ts, uint32_t is, uint32_t so, uint32_t rm, uint32_t ms), (ts, is, so, rm, ms));
|
||||
|
||||
extern "C" void GX__SetTevIndirect_801717ac(uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as) {
|
||||
static void GX__SetTevIndirect_801717ac_gx(uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as) {
|
||||
GXSetTevIndirect((GXTevStageID)std::min(ts,15u), (GXIndTexStageID)std::min(is,3u), (GXIndTexFormat)f,
|
||||
(GXIndTexBiasSel)bs, (GXIndTexMtxID)ms, (GXIndTexWrap)ws, (GXIndTexWrap)wt,
|
||||
(GXBool)ap, (GXBool)il, (GXIndTexAlphaSel)as);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801717ac, GX__SetTevIndirect_801717ac, (uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as), (ts, is, f, bs, ms, ws, wt, ap, il, as));
|
||||
GX_DEFERRED_OVERRIDE_VOID(801717ac, GX__SetTevIndirect_801717ac, (uint32_t ts, uint32_t is, uint32_t f, uint32_t bs, uint32_t ms, uint32_t ws, uint32_t wt, uint32_t ap, uint32_t il, uint32_t as), (ts, is, f, bs, ms, ws, wt, ap, il, as));
|
||||
@@ -13,6 +13,9 @@ extern "C" void GX__InitFifoBase_8016c7c8(uint32_t fa, uint32_t ba, uint32_t s);
|
||||
extern "C" void GX__SetGPFifo_8016cb2c(uint32_t fa);
|
||||
extern "C" GXFifoObj* GXInit(void* base, u32 size);
|
||||
|
||||
static void GX__Init_gx(uint32_t fifoBase, uint32_t fifoSize) { GXInit(GuestToHostPtr(fifoBase, fifoSize), fifoSize); }
|
||||
static void GX__Flush_gx() { GXFlush(); }
|
||||
|
||||
// ============================================================================
|
||||
// GXInit
|
||||
// ============================================================================
|
||||
@@ -28,7 +31,7 @@ extern "C" uint32_t GX__Init_8016b850(uint32_t fifoBase, uint32_t fifoSize)
|
||||
constexpr uint32_t kGXDataAddr = 0x803437C0u;
|
||||
constexpr uint32_t kGXDataSize = 0x600u;
|
||||
|
||||
GXInit(GuestToHostPtr(fifoBase, fifoSize), fifoSize);
|
||||
GxThread::Post(&GX__Init_gx, fifoBase, fifoSize);
|
||||
|
||||
// Initialize GXData structure in guest memory
|
||||
try {
|
||||
@@ -89,19 +92,20 @@ PPC_NATIVE_OVERRIDE(8016b850, GX__Init_8016b850, uint32_t, (uint32_t fifoBase, u
|
||||
// FIFO Management
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__InitFifoBase_8016c7c8(uint32_t fa, uint32_t ba, uint32_t s) { GXInitFifoBase((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)), GuestToHostPtr(ba, s), s); }
|
||||
static void GX__InitFifoBase_8016c7c8_gx(uint32_t fa, uint32_t ba, uint32_t s) { GXInitFifoBase((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)), GuestToHostPtr(ba, s), s); }
|
||||
extern "C" void GX__InitFifoBase_8016c7c8(uint32_t fa, uint32_t ba, uint32_t s) { GxThread::Post(&GX__InitFifoBase_8016c7c8_gx, fa, ba, s); }
|
||||
|
||||
extern "C" void GX__SetCPUFifo_8016c94c(uint32_t fa) { GXSetCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016c94c, GX__SetCPUFifo_8016c94c, (uint32_t fa), (fa));
|
||||
static void GX__SetCPUFifo_8016c94c_gx(uint32_t fa) { GXSetCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016c94c, GX__SetCPUFifo_8016c94c, (uint32_t fa), (fa));
|
||||
|
||||
extern "C" void GX__SetGPFifo_8016cb2c(uint32_t fa) { GXSetGPFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016cb2c, GX__SetGPFifo_8016cb2c, (uint32_t fa), (fa));
|
||||
static void GX__SetGPFifo_8016cb2c_gx(uint32_t fa) { GXSetGPFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016cb2c, GX__SetGPFifo_8016cb2c, (uint32_t fa), (fa));
|
||||
|
||||
extern "C" void __GX__SaveFifo_8016cdbc(uint32_t fa) { GXSaveCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016cdbc, __GX__SaveFifo_8016cdbc, (uint32_t fa), (fa));
|
||||
static void __GX__SaveFifo_8016cdbc_gx(uint32_t fa) { GXSaveCPUFifo((GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj))); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016cdbc, __GX__SaveFifo_8016cdbc, (uint32_t fa), (fa));
|
||||
|
||||
extern "C" void GX__GetCPUFifo_8016cf10(uint32_t fa) { auto* d=(GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)); auto* s=GXGetCPUFifo(); if(d&&s) std::memcpy(d,s,sizeof(GXFifoObj)); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016cf10, GX__GetCPUFifo_8016cf10, (uint32_t fa), (fa));
|
||||
static void GX__GetCPUFifo_8016cf10_gx(uint32_t fa) { auto* d=(GXFifoObj*)GuestToHostPtr(fa, sizeof(GXFifoObj)); auto* s=GXGetCPUFifo(); if(d&&s) std::memcpy(d,s,sizeof(GXFifoObj)); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016cf10, GX__GetCPUFifo_8016cf10, (uint32_t fa), (fa));
|
||||
|
||||
extern "C" void __GX__FifoInit_8016d180()
|
||||
{
|
||||
@@ -181,7 +185,7 @@ extern "C" void GX__BeginDisplayList_80172e00(uint32_t la, uint32_t s) {
|
||||
// Mirror the guest fifo-object fields the FIFO write path consumes so
|
||||
// HleFifoWrite never has to read them back out of guest memory.
|
||||
BeginDisplayListRecording(la, s);
|
||||
GXFlush();
|
||||
GxThread::Post(&GX__Flush_gx);
|
||||
GX__GetCPUFifo_8016cf10(0x80344710);
|
||||
GX__SetCPUFifo_8016c94c(0x80344090);
|
||||
} catch (...) {}
|
||||
@@ -190,7 +194,7 @@ PPC_NATIVE_OVERRIDE_VOID(80172e00, GX__BeginDisplayList_80172e00, (uint32_t la,
|
||||
|
||||
extern "C" uint32_t GX__EndDisplayList_80172eb4() {
|
||||
try {
|
||||
GXFlush();
|
||||
GxThread::Post(&GX__Flush_gx);
|
||||
GX__GetCPUFifo_8016cf10(0x80344090);
|
||||
const uint8_t wrapped = Memory::Read8(kDlFifoAddr + kDlWrapFlagOffset);
|
||||
GX__SetCPUFifo_8016c94c(0x80344710);
|
||||
@@ -229,5 +233,5 @@ PPC_NATIVE_OVERRIDE(80172EB4, GX__EndDisplayList_80172eb4, uint32_t, (), ());
|
||||
// Flush
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__Flush_8016e654() { GXFlush(); }
|
||||
extern "C" void GX__Flush_8016e654() { GxThread::Post(&GX__Flush_gx); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016e654, GX__Flush_8016e654, (), ());
|
||||
@@ -3,6 +3,7 @@
|
||||
#include "hle_stubs.h"
|
||||
#include "memory.h"
|
||||
#include "gx_guest_write.h"
|
||||
#include "gx_thread.h"
|
||||
#include "ppc_runtime.h"
|
||||
#include "aurora_events.h"
|
||||
#include "gx_texture_binding_contract.h"
|
||||
@@ -72,6 +73,7 @@ extern "C" uint32_t OS__GetCurrentThread_801a98b0_hle();
|
||||
extern "C" void GX__SetCPUFifo_8016c94c(uint32_t fifoAddr);
|
||||
extern "C" void GX__SetDirtyState_8016ee78();
|
||||
extern "C" void GX__CallDisplayList_80172f64(uint32_t listAddr, uint32_t nbytes);
|
||||
void GX__CallDisplayList_gx(uint32_t listAddr, uint32_t nbytes);
|
||||
|
||||
extern std::atomic_bool g_auroraFrameActive;
|
||||
extern std::atomic_bool g_auroraFrameHadWork;
|
||||
@@ -141,8 +143,6 @@ struct TexObjSlot : TexObjMeta {
|
||||
// index vectors can carry slot pointers instead of re-looking-up keys.
|
||||
uint32_t objAddr = 0;
|
||||
|
||||
// Aurora-side object. Created lazily by CreateHostTexObj.
|
||||
std::unique_ptr<HleTexObj> host;
|
||||
|
||||
// Byte-exact shadow of the last-decoded guest GXTexObj; served without a
|
||||
// diff only while guest bytes still match it. Games mutate these structs
|
||||
@@ -343,18 +343,11 @@ TexObjMeta& GetTexObjMeta(uint32_t addr);
|
||||
TexObjMeta ExtractTexObjMetaFromGuest(uint32_t addr);
|
||||
bool TryGetOrExtractTexObjMeta(uint32_t addr, TexObjMeta& outMeta);
|
||||
TlutObjMeta& GetTlutObjMeta(uint32_t addr);
|
||||
GXTexObj* GetHostTexObj(uint32_t addr);
|
||||
GXTexObj* TryGetHostTexObj(uint32_t addr);
|
||||
GXTexObj* CreateHostTexObj(uint32_t addr);
|
||||
void MarkHostTexObjConstructed(uint32_t addr);
|
||||
void MarkTexObjsDirtyForRange(uint32_t addr, uint32_t size);
|
||||
// Invalidates GPU-only GXCopyTex results when guest CPU writes are made visible
|
||||
// over their destination. This is the allocation-reuse generation boundary
|
||||
// for copy textures; it must be called for every data-cache store/flush range.
|
||||
void InvalidateEfbCopyDestinationsForRange(uint32_t addr, uint32_t size);
|
||||
GXTlutObj* CreateHostTlutObj(uint32_t addr);
|
||||
void MarkHostTlutObjConstructed(uint32_t addr);
|
||||
GXTlutObj* GetHostTlutObj(uint32_t addr);
|
||||
void MarkTlutObjsDirtyForRange(uint32_t addr, uint32_t size);
|
||||
// One-stop invalidation for DMA-class host-side writes into guest RAM (DVD
|
||||
// reads, DCZeroRange, LC stores): texobjs + TLUTs + EFB copies + display
|
||||
@@ -363,6 +356,50 @@ void MarkTlutObjsDirtyForRange(uint32_t addr, uint32_t size);
|
||||
extern "C" void GxNotifyGuestRamDmaWrite(uint32_t addr, uint32_t size);
|
||||
|
||||
void HleFifoWrite(u32 val, uint32_t sizeBytes);
|
||||
// GX-thread side of the write-gather pipe: parses a run of FIFO bytes.
|
||||
extern "C" void GxFifoConsumeBytes(const uint8_t* data, uint32_t sizeBytes);
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// GX thread split. GX_DEFERRED_OVERRIDE_VOID registers a guest-facing front
|
||||
// that posts `name_gx` (the aurora work) to the GX thread; see gx_thread.h.
|
||||
// ---------------------------------------------------------------------------
|
||||
#define GX_COMMA_ARGS(...) , ##__VA_ARGS__
|
||||
#define GX_DEFERRED_OVERRIDE_VOID(addr_hex, name, arg_list, call_list) \
|
||||
extern "C" void name arg_list { GxThread::Post(&name##_gx GX_COMMA_ARGS call_list); } \
|
||||
PPC_NATIVE_OVERRIDE_VOID(addr_hex, name, arg_list, call_list)
|
||||
|
||||
// Game-thread mirrors of the vertex descriptor state, kept for the GXGet*
|
||||
// overrides; the GX thread owns g_hleGxState.
|
||||
struct GxGameSideVertexState {
|
||||
GXAttrType vtxDesc[26]{};
|
||||
VtxAttrFmt vtxAttrFmt[8][26]{};
|
||||
};
|
||||
extern GxGameSideVertexState g_gxGameVertexState;
|
||||
|
||||
// Texture objects: the game thread keeps the meta table (guest shadows, dirty
|
||||
// flags, getters); the GX thread keeps the aurora objects and rebuilds them
|
||||
// from the meta snapshot each load carries.
|
||||
struct GxTexObjLoad {
|
||||
TexObjMeta meta;
|
||||
uint32_t objAddr = 0;
|
||||
uint32_t tid = 0;
|
||||
bool upload = false;
|
||||
};
|
||||
void GxHostLoadTexObj_gx(GxTexObjLoad load);
|
||||
void GxHostBindPlaceholder_gx(uint32_t tid);
|
||||
struct GxTlutLoad {
|
||||
TlutObjMeta meta;
|
||||
uint32_t objAddr = 0;
|
||||
uint32_t tlut = 0;
|
||||
bool rebuild = false;
|
||||
bool valid = false;
|
||||
};
|
||||
void GxHostLoadTlut_gx(GxTlutLoad load);
|
||||
void GxHostDestroyCopyTex_gx(uint32_t copyAddr);
|
||||
// Backs shared by the EGG helpers.
|
||||
void GX__SetTevColor_gx(uint32_t id, uint32_t colorWord);
|
||||
void GX__SetTevKColor_gx(uint32_t id, uint32_t colorWord);
|
||||
void GX__Begin_gx(uint32_t t, uint32_t vf, uint32_t nv);
|
||||
void SubmitAttribute(GXAttr attr, float* comps, const VtxAttrFmt& fmt, const u32* rawComps = nullptr);
|
||||
void SubmitIndexedAttribute(GXAttr attr, uint32_t index);
|
||||
|
||||
|
||||
@@ -27,12 +27,29 @@ bool ValidateGuestLightObj(uint32_t addr, const char* operation) {
|
||||
return false;
|
||||
}
|
||||
|
||||
float ReadGuestLightFloat(uint32_t addr, uint32_t offset) {
|
||||
return Memory::ReadFloat32(addr + offset);
|
||||
// The SDK copies the light object into the FIFO at call time, so the guest
|
||||
// object is snapshotted on the game thread and decoded on the GX thread.
|
||||
struct GuestLightSnapshot {
|
||||
uint32_t words[kLightObjSize / 4];
|
||||
};
|
||||
|
||||
GuestLightSnapshot SnapshotGuestLight(uint32_t addr) {
|
||||
GuestLightSnapshot snapshot{};
|
||||
for (uint32_t i = 0; i < kLightObjSize / 4; ++i) {
|
||||
snapshot.words[i] = Memory::Read32(addr + i * 4);
|
||||
}
|
||||
return snapshot;
|
||||
}
|
||||
|
||||
void InitializeHostLightFromGuest(GXLightObj& host, uint32_t guestAddr) {
|
||||
GXInitLightColor(&host, DecodeGxColor(Memory::Read32(guestAddr + kColorOffset)));
|
||||
float ReadGuestLightFloat(const GuestLightSnapshot& light, uint32_t offset) {
|
||||
float value;
|
||||
const uint32_t bits = light.words[offset / 4];
|
||||
std::memcpy(&value, &bits, sizeof(value));
|
||||
return value;
|
||||
}
|
||||
|
||||
void InitializeHostLightFromGuest(GXLightObj& host, const GuestLightSnapshot& guestAddr) {
|
||||
GXInitLightColor(&host, DecodeGxColor(guestAddr.words[kColorOffset / 4]));
|
||||
GXInitLightAttn(
|
||||
&host,
|
||||
ReadGuestLightFloat(guestAddr, kAttnAOffset + 0),
|
||||
@@ -62,11 +79,14 @@ void InitializeHostLightFromGuest(GXLightObj& host, uint32_t guestAddr) {
|
||||
// Light Object Load
|
||||
// ============================================================================
|
||||
|
||||
static void GX__LoadLightObjImm_gx(GuestLightSnapshot light, uint32_t lid) {
|
||||
GXLightObj host{};
|
||||
InitializeHostLightFromGuest(host, light);
|
||||
GXLoadLightObjImm(&host, static_cast<GXLightID>(lid));
|
||||
}
|
||||
extern "C" void GX__LoadLightObjImm_80170320(uint32_t la, uint32_t lid) {
|
||||
if (!ValidateGuestLightObj(la, "GXLoadLightObjImm")) return;
|
||||
GXLightObj host{};
|
||||
InitializeHostLightFromGuest(host, la);
|
||||
GXLoadLightObjImm(&host, static_cast<GXLightID>(lid));
|
||||
GxThread::Post(&GX__LoadLightObjImm_gx, SnapshotGuestLight(la), lid);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170320, GX__LoadLightObjImm_80170320, (uint32_t la, uint32_t lid), (la, lid));
|
||||
|
||||
@@ -74,24 +94,28 @@ PPC_NATIVE_OVERRIDE_VOID(80170320, GX__LoadLightObjImm_80170320, (uint32_t la, u
|
||||
// Channel Control
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetChanAmbColor_8017039c(uint32_t c, uint32_t cp) {
|
||||
static void GX__SetChanAmbColor_gx(uint32_t c, uint32_t colorWord) {
|
||||
EnsureAuroraFrameActive();
|
||||
GXColor color = DecodeGxColor(Memory::Read32(cp));
|
||||
GXSetChanAmbColor((GXChannelID)c, color);
|
||||
GXSetChanAmbColor((GXChannelID)c, DecodeGxColor(colorWord));
|
||||
}
|
||||
extern "C" void GX__SetChanAmbColor_8017039c(uint32_t c, uint32_t cp) {
|
||||
GxThread::Post(&GX__SetChanAmbColor_gx, c, Memory::Read32(cp));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017039c, GX__SetChanAmbColor_8017039c, (uint32_t c, uint32_t cp), (c, cp));
|
||||
|
||||
extern "C" void GX__SetChanMatColor_80170474(uint32_t c, uint32_t cp) {
|
||||
static void GX__SetChanMatColor_gx(uint32_t c, uint32_t colorWord) {
|
||||
EnsureAuroraFrameActive();
|
||||
GXColor color = DecodeGxColor(Memory::Read32(cp));
|
||||
GXSetChanMatColor((GXChannelID)c, color);
|
||||
GXSetChanMatColor((GXChannelID)c, DecodeGxColor(colorWord));
|
||||
}
|
||||
extern "C" void GX__SetChanMatColor_80170474(uint32_t c, uint32_t cp) {
|
||||
GxThread::Post(&GX__SetChanMatColor_gx, c, Memory::Read32(cp));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170474, GX__SetChanMatColor_80170474, (uint32_t c, uint32_t cp), (c, cp));
|
||||
|
||||
extern "C" void GX__SetNumChans_8017054c(uint32_t n) { GXSetNumChans((u8)n); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017054c, GX__SetNumChans_8017054c, (uint32_t n), (n));
|
||||
static void GX__SetNumChans_8017054c_gx(uint32_t n) { GXSetNumChans((u8)n); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017054c, GX__SetNumChans_8017054c, (uint32_t n), (n));
|
||||
|
||||
extern "C" void GX__SetChanCtrl_80170570(uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af) {
|
||||
static void GX__SetChanCtrl_80170570_gx(uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af) {
|
||||
GXSetChanCtrl((GXChannelID)ch, en!=0, (GXColorSrc)as, (GXColorSrc)ms, lm, (GXDiffuseFn)df, (GXAttnFn)af);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170570, GX__SetChanCtrl_80170570, (uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af), (ch, en, as, ms, lm, df, af));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80170570, GX__SetChanCtrl_80170570, (uint32_t ch, uint32_t en, uint32_t as, uint32_t ms, uint32_t lm, uint32_t df, uint32_t af), (ch, en, as, ms, lm, df, af));
|
||||
@@ -15,48 +15,9 @@ using TextureHandle = std::shared_ptr<TextureRef>;
|
||||
// HleTexObj now lives in gx_internal.h: it is embedded in TexObjSlot so the
|
||||
// metadata and the Aurora object share one hash-map entry.
|
||||
|
||||
struct HostGXTlutObj {
|
||||
alignas(GXTlutObj) std::byte publicStorage[sizeof(GXTlutObj)]{};
|
||||
aurora::gfx::TextureHandle ref;
|
||||
};
|
||||
|
||||
struct HleTlutObj {
|
||||
static constexpr size_t kTlutObjStorageSize =
|
||||
(sizeof(GXTlutObj) >= sizeof(HostGXTlutObj)) ? sizeof(GXTlutObj) : sizeof(HostGXTlutObj);
|
||||
static constexpr size_t kTlutObjStorageAlign =
|
||||
(alignof(GXTlutObj) >= alignof(HostGXTlutObj)) ? alignof(GXTlutObj) : alignof(HostGXTlutObj);
|
||||
using Storage = std::aligned_storage_t<kTlutObjStorageSize, kTlutObjStorageAlign>;
|
||||
|
||||
Storage storage{};
|
||||
bool storageLive = false;
|
||||
bool constructed = false;
|
||||
|
||||
HleTlutObj() = default;
|
||||
~HleTlutObj() { Destroy(); }
|
||||
|
||||
HostGXTlutObj* HostObj() { return reinterpret_cast<HostGXTlutObj*>(&storage); }
|
||||
GXTlutObj* PublicPtr() { return reinterpret_cast<GXTlutObj*>(HostObj()->publicStorage); }
|
||||
|
||||
void EnsureStorageLive() {
|
||||
if (!storageLive) {
|
||||
new (&storage) HostGXTlutObj();
|
||||
storageLive = true;
|
||||
}
|
||||
}
|
||||
|
||||
void Destroy() {
|
||||
if (storageLive) {
|
||||
GXDestroyTlutObj(PublicPtr());
|
||||
std::destroy_at(HostObj());
|
||||
std::memset(&storage, 0, sizeof(storage));
|
||||
storageLive = false;
|
||||
constructed = false;
|
||||
}
|
||||
}
|
||||
};
|
||||
// The aurora TLUT objects live on the GX thread (gx_texture.cpp).
|
||||
|
||||
std::mutex g_texObjMutex;
|
||||
std::map<uint32_t, std::unique_ptr<HleTlutObj>> g_HostTlutObjMap;
|
||||
std::mutex g_tlutObjMutex;
|
||||
|
||||
// The texobj table. One slot per guest GXTexObj address holds both the decoded
|
||||
@@ -548,52 +509,6 @@ TlutObjMeta& GetTlutObjMeta(uint32_t addr) {
|
||||
return g_TlutObjMeta[addr];
|
||||
}
|
||||
|
||||
GXTexObj* GetHostTexObj(uint32_t addr) {
|
||||
TexObjSlot* slot = FindTexObjSlot(addr);
|
||||
if (slot == nullptr || !slot->host || !slot->host->constructed) {
|
||||
const CpuContext* cpu = TryGetCpuContext();
|
||||
RT_LOGF(RT_TAG_GX,
|
||||
"GXTex: invalid GXTexObj @0x%08X (PC=0x%08X, LR=0x%08X, CTR=0x%08X)\n",
|
||||
addr, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u, cpu ? cpu->ctr : 0u);
|
||||
std::fflush(stderr);
|
||||
GXTexObj* obj = CreateHostTexObj(addr);
|
||||
MarkHostTexObjConstructed(addr);
|
||||
return obj;
|
||||
}
|
||||
return slot->host->PublicPtr();
|
||||
}
|
||||
|
||||
GXTexObj* CreateHostTexObj(uint32_t addr) {
|
||||
TexObjSlot& slot = FindOrCreateTexObjSlot(addr);
|
||||
// GXInitTexObj/GXInitTexObjCI land here and rewrite the entire guest struct,
|
||||
// resetting the LOD/filter words the HLE does not mirror into this cache.
|
||||
// The shadow captured before that rewrite mismatches afterwards, so the
|
||||
// next lookup takes exactly one guest re-decode and picks those defaults
|
||||
// (and the computed mipmap maxLod) up.
|
||||
if (!slot.host) {
|
||||
slot.host = std::make_unique<HleTexObj>();
|
||||
} else {
|
||||
slot.host->Destroy();
|
||||
}
|
||||
slot.host->EnsureStorageLive();
|
||||
return slot.host->PublicPtr();
|
||||
}
|
||||
|
||||
void MarkHostTexObjConstructed(uint32_t addr) {
|
||||
TexObjSlot* slot = FindTexObjSlot(addr);
|
||||
if (slot != nullptr && slot->host) {
|
||||
slot->host->constructed = true;
|
||||
}
|
||||
}
|
||||
|
||||
GXTexObj* TryGetHostTexObj(uint32_t addr) {
|
||||
TexObjSlot* slot = FindTexObjSlot(addr);
|
||||
if (slot != nullptr && slot->host && slot->host->constructed) {
|
||||
return slot->host->PublicPtr();
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
void MarkTexObjsDirtyForRange(uint32_t addr, uint32_t size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
@@ -656,39 +571,6 @@ extern "C" void GxNotifyGuestRamDmaWrite(uint32_t addr, uint32_t size) {
|
||||
GxNotifyDisplayListMemoryWrite(addr, size);
|
||||
}
|
||||
|
||||
GXTlutObj* CreateHostTlutObj(uint32_t addr) {
|
||||
auto& slot = g_HostTlutObjMap[addr];
|
||||
if (!slot) {
|
||||
slot = std::make_unique<HleTlutObj>();
|
||||
} else {
|
||||
slot->Destroy();
|
||||
}
|
||||
slot->EnsureStorageLive();
|
||||
return slot->PublicPtr();
|
||||
}
|
||||
|
||||
void MarkHostTlutObjConstructed(uint32_t addr) {
|
||||
auto it = g_HostTlutObjMap.find(addr);
|
||||
if (it != g_HostTlutObjMap.end() && it->second) {
|
||||
it->second->constructed = true;
|
||||
}
|
||||
}
|
||||
|
||||
GXTlutObj* GetHostTlutObj(uint32_t addr) {
|
||||
auto it = g_HostTlutObjMap.find(addr);
|
||||
if (it == g_HostTlutObjMap.end() || !it->second || !it->second->constructed) {
|
||||
const CpuContext* cpu = TryGetCpuContext();
|
||||
RT_LOGF(RT_TAG_GX,
|
||||
"GXTex: invalid GXTlutObj @0x%08X (PC=0x%08X, LR=0x%08X, CTR=0x%08X)\n",
|
||||
addr, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u, cpu ? cpu->ctr : 0u);
|
||||
std::fflush(stderr);
|
||||
GXTlutObj* obj = CreateHostTlutObj(addr);
|
||||
MarkHostTlutObjConstructed(addr);
|
||||
return obj;
|
||||
}
|
||||
return it->second->PublicPtr();
|
||||
}
|
||||
|
||||
void MarkTlutObjsDirtyForRange(uint32_t addr, uint32_t size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
|
||||
@@ -5,84 +5,88 @@
|
||||
// Blend Mode
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetBlendMode_8017277c(uint32_t t, uint32_t s, uint32_t d, uint32_t op) {
|
||||
static void GX__SetBlendMode_8017277c_gx(uint32_t t, uint32_t s, uint32_t d, uint32_t op) {
|
||||
GXSetBlendMode(static_cast<GXBlendMode>(t), static_cast<GXBlendFactor>(s),
|
||||
static_cast<GXBlendFactor>(d), static_cast<GXLogicOp>(op));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017277c, GX__SetBlendMode_8017277c, (uint32_t t, uint32_t s, uint32_t d, uint32_t op), (t, s, d, op));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017277c, GX__SetBlendMode_8017277c, (uint32_t t, uint32_t s, uint32_t d, uint32_t op), (t, s, d, op));
|
||||
|
||||
extern "C" void GX__SetColorUpdate_801727cc(uint32_t en) {
|
||||
static void GX__SetColorUpdate_801727cc_gx(uint32_t en) {
|
||||
GXSetColorUpdate(static_cast<GXBool>(en));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801727cc, GX__SetColorUpdate_801727cc, (uint32_t en), (en));
|
||||
GX_DEFERRED_OVERRIDE_VOID(801727cc, GX__SetColorUpdate_801727cc, (uint32_t en), (en));
|
||||
|
||||
extern "C" void GX__SetAlphaUpdate_801727f8(uint32_t en) {
|
||||
static void GX__SetAlphaUpdate_801727f8_gx(uint32_t en) {
|
||||
GXSetAlphaUpdate(static_cast<GXBool>(en));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801727f8, GX__SetAlphaUpdate_801727f8, (uint32_t en), (en));
|
||||
GX_DEFERRED_OVERRIDE_VOID(801727f8, GX__SetAlphaUpdate_801727f8, (uint32_t en), (en));
|
||||
|
||||
// ============================================================================
|
||||
// Z Buffer
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetZMode_80172824(uint32_t ce, uint32_t f, uint32_t ue) {
|
||||
static void GX__SetZMode_80172824_gx(uint32_t ce, uint32_t f, uint32_t ue) {
|
||||
GXSetZMode(static_cast<GXBool>(ce), static_cast<GXCompare>(f), static_cast<GXBool>(ue));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172824, GX__SetZMode_80172824, (uint32_t ce, uint32_t f, uint32_t ue), (ce, f, ue));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80172824, GX__SetZMode_80172824, (uint32_t ce, uint32_t f, uint32_t ue), (ce, f, ue));
|
||||
|
||||
extern "C" void GX__SetZCompLoc_80172858(uint32_t bt) {
|
||||
static void GX__SetZCompLoc_80172858_gx(uint32_t bt) {
|
||||
GXSetZCompLoc(static_cast<GXBool>(bt));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172858, GX__SetZCompLoc_80172858, (uint32_t bt), (bt));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80172858, GX__SetZCompLoc_80172858, (uint32_t bt), (bt));
|
||||
|
||||
// ============================================================================
|
||||
// Pixel Format and Dither
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetPixelFmt_80172888(uint32_t pf, uint32_t zf) {
|
||||
static void GX__SetPixelFmt_80172888_gx(uint32_t pf, uint32_t zf) {
|
||||
GXSetPixelFmt(static_cast<GXPixelFmt>(pf), static_cast<GXZFmt16>(zf));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172888, GX__SetPixelFmt_80172888, (uint32_t pf, uint32_t zf), (pf, zf));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80172888, GX__SetPixelFmt_80172888, (uint32_t pf, uint32_t zf), (pf, zf));
|
||||
|
||||
extern "C" void GX__SetDither_80172930(uint32_t d) {
|
||||
static void GX__SetDither_80172930_gx(uint32_t d) {
|
||||
GXSetDither(static_cast<GXBool>(d));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172930, GX__SetDither_80172930, (uint32_t d), (d));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80172930, GX__SetDither_80172930, (uint32_t d), (d));
|
||||
|
||||
extern "C" void GX__SetDstAlpha_8017295c(uint32_t en, uint32_t a) {
|
||||
static void GX__SetDstAlpha_8017295c_gx(uint32_t en, uint32_t a) {
|
||||
GXSetDstAlpha(static_cast<GXBool>(en), static_cast<u8>(a));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017295c, GX__SetDstAlpha_8017295c, (uint32_t en, uint32_t a), (en, a));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017295c, GX__SetDstAlpha_8017295c, (uint32_t en, uint32_t a), (en, a));
|
||||
|
||||
// ============================================================================
|
||||
// Fog and Alpha Compare
|
||||
// ============================================================================
|
||||
|
||||
static void GX__SetFog_gx(uint32_t t, float sz, float ez, float nz, float fz, uint32_t colorWord) {
|
||||
GXSetFog((GXFogType)t, sz, ez, nz, fz, DecodeGxColor(colorWord));
|
||||
}
|
||||
extern "C" void GX__SetFog_801722cc(uint32_t t, float sz, float ez, float nz, float fz, uint32_t cp) {
|
||||
GXSetFog((GXFogType)t, sz, ez, nz, fz, DecodeGxColor(Memory::Read32(cp)));
|
||||
// The colour is copied out of guest memory at call time, as the SDK does.
|
||||
GxThread::Post(&GX__SetFog_gx, t, sz, ez, nz, fz, Memory::Read32(cp));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801722cc, GX__SetFog_801722cc, (uint32_t t, float sz, float ez, float nz, float fz, uint32_t cp), (t, sz, ez, nz, fz, cp));
|
||||
|
||||
extern "C" void GX__SetAlphaCompare_80172088(uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1) {
|
||||
static void GX__SetAlphaCompare_80172088_gx(uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1) {
|
||||
g_alphaCompareValid = true;
|
||||
GXSetAlphaCompare((GXCompare)c0, (u8)r0, (GXAlphaOp)op, (GXCompare)c1, (u8)r1);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172088, GX__SetAlphaCompare_80172088, (uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1), (c0, r0, op, c1, r1));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80172088, GX__SetAlphaCompare_80172088, (uint32_t c0, uint32_t r0, uint32_t op, uint32_t c1, uint32_t r1), (c0, r0, op, c1, r1));
|
||||
|
||||
extern "C" void GX__SetZTexture_801720c0(uint32_t op, uint32_t f, uint32_t b) { GXSetZTexture((GXZTexOp)op, (GXTexFmt)f, b); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(801720c0, GX__SetZTexture_801720c0, (uint32_t op, uint32_t f, uint32_t b), (op, f, b));
|
||||
static void GX__SetZTexture_801720c0_gx(uint32_t op, uint32_t f, uint32_t b) { GXSetZTexture((GXZTexOp)op, (GXTexFmt)f, b); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(801720c0, GX__SetZTexture_801720c0, (uint32_t op, uint32_t f, uint32_t b), (op, f, b));
|
||||
|
||||
// ============================================================================
|
||||
// Culling and Clipping
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetCullMode_8016f3b8(uint32_t m) {
|
||||
static void GX__SetCullMode_8016f3b8_gx(uint32_t m) {
|
||||
GXSetCullMode(static_cast<GXCullMode>(m));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f3b8, GX__SetCullMode_8016f3b8, (uint32_t m), (m));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f3b8, GX__SetCullMode_8016f3b8, (uint32_t m), (m));
|
||||
|
||||
extern "C" void GX__SetCoPlanar_8016f3e0(uint32_t en) { GXSetCoPlanar((GXBool)en); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f3e0, GX__SetCoPlanar_8016f3e0, (uint32_t en), (en));
|
||||
static void GX__SetCoPlanar_8016f3e0_gx(uint32_t en) { GXSetCoPlanar((GXBool)en); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f3e0, GX__SetCoPlanar_8016f3e0, (uint32_t en), (en));
|
||||
|
||||
extern "C" void GX__SetClipMode_8017351c(uint32_t m) { GXSetClipMode((GXClipMode)m); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017351c, GX__SetClipMode_8017351c, (uint32_t m), (m));
|
||||
static void GX__SetClipMode_8017351c_gx(uint32_t m) { GXSetClipMode((GXClipMode)m); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017351c, GX__SetClipMode_8017351c, (uint32_t m), (m));
|
||||
@@ -3,18 +3,7 @@
|
||||
|
||||
extern "C" void __GXSetSUTexRegs();
|
||||
|
||||
// ============================================================================
|
||||
// FIFO Write Helpers
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX_HLE_FIFO_WriteFloat(float val) {
|
||||
u32 raw; std::memcpy(&raw, &val, 4);
|
||||
try { HleFifoWrite(raw, 4); } catch (...) { RT_LOGF(RT_TAG_GX, "FIFO write float failed\n"); }
|
||||
}
|
||||
|
||||
extern "C" void GX_HLE_FIFO_Write32(uint32_t val) { HleFifoWrite(val, 4); }
|
||||
extern "C" void GX_HLE_FIFO_Write16(uint16_t val) { HleFifoWrite(static_cast<u32>(val), 2); }
|
||||
extern "C" void GX_HLE_FIFO_Write8(uint8_t val) { HleFifoWrite(static_cast<u32>(val), 1); }
|
||||
// The FIFO write helpers (GX_HLE_FIFO_Write*) live in gx_fifo.cpp.
|
||||
|
||||
extern "C" void GX__SetDrawSync_8016ed08(uint32_t token) {
|
||||
(void)token;
|
||||
@@ -36,15 +25,21 @@ extern "C" void GX__FinishInterruptHandler_8016ed94() {
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016ed94, GX__FinishInterruptHandler_8016ed94, (), ());
|
||||
|
||||
static void GX__DrawDone_gx() { GXDrawDone(); }
|
||||
extern "C" void GX__DrawDone_8016eab0() {
|
||||
try { Memory::Write8(kGxDrawDoneFlagAddr, 0); } catch (...) {}
|
||||
GXDrawDone(); GX__FinishInterruptHandler_8016ed94();
|
||||
// Hardware blocks here until the GP has consumed the FIFO: post the drain
|
||||
// and wait for the GX thread to reach it before raising the finish flags.
|
||||
GxThread::Post(&GX__DrawDone_gx);
|
||||
GxThread::Drain();
|
||||
GX__FinishInterruptHandler_8016ed94();
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016eab0, GX__DrawDone_8016eab0, (), ());
|
||||
|
||||
static void GX__PixModeSync_gx() { GXPixModeSync(); }
|
||||
extern "C" void GX__PixModeSync_8016eb70() {
|
||||
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {}
|
||||
GXPixModeSync();
|
||||
GxThread::Post(&GX__PixModeSync_gx);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016eb70, GX__PixModeSync_8016eb70, (), ());
|
||||
|
||||
@@ -59,8 +54,9 @@ PPC_NATIVE_OVERRIDE_VOID(8016b720, __GX__InitRevisionBits_8016b720, (), ());
|
||||
// Texture State Management - Aurora handles internally
|
||||
// ============================================================================
|
||||
|
||||
static void __GX__SetSUTexRegs_gx() { __GXSetSUTexRegs(); }
|
||||
extern "C" void __GX__SetSUTexRegs_801712f0() {
|
||||
__GXSetSUTexRegs();
|
||||
GxThread::Post(&__GX__SetSUTexRegs_gx);
|
||||
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801712f0, __GX__SetSUTexRegs_801712f0, (), ());
|
||||
@@ -81,8 +77,9 @@ PPC_NATIVE_OVERRIDE_VOID(80171c28, __GX__FlushTextureState_80171c28, (), ());
|
||||
// Copy Configuration - No-ops for features Aurora doesn't use
|
||||
// ============================================================================
|
||||
|
||||
static void GX__SetDispCopyFrame2Field_gx(uint32_t f) { GXSetDispCopyFrame2Field(f); }
|
||||
extern "C" void GX__SetDispCopyFrame2Field_8016f5f8(uint32_t f) {
|
||||
GXSetDispCopyFrame2Field(f);
|
||||
GxThread::Post(&GX__SetDispCopyFrame2Field_gx, f);
|
||||
try {
|
||||
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
|
||||
if (gd) {
|
||||
@@ -93,8 +90,9 @@ extern "C" void GX__SetDispCopyFrame2Field_8016f5f8(uint32_t f) {
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f5f8, GX__SetDispCopyFrame2Field_8016f5f8, (uint32_t f), (f));
|
||||
|
||||
static void GX__SetCopyClamp_gx(uint32_t c) { GXSetCopyClamp(static_cast<GXFBClamp>(c)); }
|
||||
extern "C" void GX__SetCopyClamp_8016f618(uint32_t c) {
|
||||
GXSetCopyClamp(static_cast<GXFBClamp>(c));
|
||||
GxThread::Post(&GX__SetCopyClamp_gx, c);
|
||||
try {
|
||||
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
|
||||
if (gd) {
|
||||
@@ -106,8 +104,9 @@ extern "C" void GX__SetCopyClamp_8016f618(uint32_t c) {
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f618, GX__SetCopyClamp_8016f618, (uint32_t c), (c));
|
||||
|
||||
static void GX__ClearBoundingBox_gx() { GXClearBoundingBox(); }
|
||||
extern "C" void GX__ClearBoundingBox_8016fecc() {
|
||||
GXClearBoundingBox();
|
||||
GxThread::Post(&GX__ClearBoundingBox_gx);
|
||||
try {
|
||||
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
|
||||
if (gd) Memory::Write16(gd + 2, 0);
|
||||
|
||||
@@ -24,7 +24,7 @@ inline bool TevSwapOk(uint32_t id) { return GxTevIdOk(id, GX_MAX_TEVSWAP, "TEV s
|
||||
// TEV Stage Count and Order
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetNumTevStages_801722a8(uint32_t n) {
|
||||
static void GX__SetNumTevStages_801722a8_gx(uint32_t n) {
|
||||
// GXSetNumTevStages takes a count, not an index, so the inclusive bound is
|
||||
// GX_MAX_TEVSTAGE itself.
|
||||
if (n > GX_MAX_TEVSTAGE) {
|
||||
@@ -33,96 +33,103 @@ extern "C" void GX__SetNumTevStages_801722a8(uint32_t n) {
|
||||
}
|
||||
GXSetNumTevStages((u8)n);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801722a8, GX__SetNumTevStages_801722a8, (uint32_t n), (n));
|
||||
GX_DEFERRED_OVERRIDE_VOID(801722a8, GX__SetNumTevStages_801722a8, (uint32_t n), (n));
|
||||
|
||||
extern "C" void GX__SetTevOp_80171c4c(uint32_t s, uint32_t m) {
|
||||
static void GX__SetTevOp_80171c4c_gx(uint32_t s, uint32_t m) {
|
||||
if (!TevStageOk(s)) return;
|
||||
GXSetTevOp((GXTevStageID)s, (GXTevMode)m);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171c4c, GX__SetTevOp_80171c4c, (uint32_t s, uint32_t m), (s, m));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171c4c, GX__SetTevOp_80171c4c, (uint32_t s, uint32_t m), (s, m));
|
||||
|
||||
extern "C" void GX__SetTevOrder_8017214c(uint32_t s, uint32_t c, uint32_t m, uint32_t col) {
|
||||
static void GX__SetTevOrder_8017214c_gx(uint32_t s, uint32_t c, uint32_t m, uint32_t col) {
|
||||
if (!TevStageOk(s)) return;
|
||||
GXSetTevOrder((GXTevStageID)s, (GXTexCoordID)c, (GXTexMapID)m, (GXChannelID)col);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017214c, GX__SetTevOrder_8017214c, (uint32_t s, uint32_t c, uint32_t m, uint32_t col), (s, c, m, col));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017214c, GX__SetTevOrder_8017214c, (uint32_t s, uint32_t c, uint32_t m, uint32_t col), (s, c, m, col));
|
||||
|
||||
// ============================================================================
|
||||
// TEV Color/Alpha Inputs
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetTevColorIn_80171ce0(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
|
||||
static void GX__SetTevColorIn_80171ce0_gx(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
|
||||
if (!TevStageOk(s)) return;
|
||||
GXSetTevColorIn((GXTevStageID)s, (GXTevColorArg)a, (GXTevColorArg)b, (GXTevColorArg)c, (GXTevColorArg)d);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171ce0, GX__SetTevColorIn_80171ce0, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171ce0, GX__SetTevColorIn_80171ce0, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
|
||||
|
||||
extern "C" void GX__SetTevAlphaIn_80171d20(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
|
||||
static void GX__SetTevAlphaIn_80171d20_gx(uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d) {
|
||||
if (!TevStageOk(s)) return;
|
||||
GXSetTevAlphaIn((GXTevStageID)s, (GXTevAlphaArg)a, (GXTevAlphaArg)b, (GXTevAlphaArg)c, (GXTevAlphaArg)d);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171d20, GX__SetTevAlphaIn_80171d20, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171d20, GX__SetTevAlphaIn_80171d20, (uint32_t s, uint32_t a, uint32_t b, uint32_t c, uint32_t d), (s, a, b, c, d));
|
||||
|
||||
// ============================================================================
|
||||
// TEV Color/Alpha Operations
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetTevColorOp_80171d60(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
|
||||
static void GX__SetTevColorOp_80171d60_gx(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
|
||||
if (!TevStageOk(s) || !TevRegOk(or_)) return;
|
||||
GXSetTevColorOp((GXTevStageID)s, (GXTevOp)op, (GXTevBias)b, (GXTevScale)sc, (GXBool)cl, (GXTevRegID)or_);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171d60, GX__SetTevColorOp_80171d60, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171d60, GX__SetTevColorOp_80171d60, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
|
||||
|
||||
extern "C" void GX__SetTevAlphaOp_80171db8(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
|
||||
static void GX__SetTevAlphaOp_80171db8_gx(uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_) {
|
||||
if (!TevStageOk(s) || !TevRegOk(or_)) return;
|
||||
GXSetTevAlphaOp((GXTevStageID)s, (GXTevOp)op, (GXTevBias)b, (GXTevScale)sc, (GXBool)cl, (GXTevRegID)or_);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171db8, GX__SetTevAlphaOp_80171db8, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171db8, GX__SetTevAlphaOp_80171db8, (uint32_t s, uint32_t op, uint32_t b, uint32_t sc, uint32_t cl, uint32_t or_), (s, op, b, sc, cl, or_));
|
||||
|
||||
// ============================================================================
|
||||
// TEV Color Registers
|
||||
// ============================================================================
|
||||
|
||||
void GX__SetTevColor_gx(uint32_t id, uint32_t colorWord) {
|
||||
GXSetTevColor((GXTevRegID)id, DecodeGxColor(colorWord));
|
||||
}
|
||||
extern "C" void GX__SetTevColor_80171e10(uint32_t id, uint32_t cp) {
|
||||
if (!TevRegOk(id)) return;
|
||||
const uint8_t* p=Memory::GetPointer(cp, 4); GXColor c; c.r=p[0]; c.g=p[1]; c.b=p[2]; c.a=p[3];
|
||||
GXSetTevColor((GXTevRegID)id, c);
|
||||
GxThread::Post(&GX__SetTevColor_gx, id, Memory::Read32(cp));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171e10, GX__SetTevColor_80171e10, (uint32_t id, uint32_t cp), (id, cp));
|
||||
|
||||
static void GX__SetTevColorS10_gx(uint32_t id, uint32_t rg, uint32_t ba) {
|
||||
GXColorS10 c;
|
||||
c.r=static_cast<s16>(rg>>16); c.g=static_cast<s16>(rg&0xFFFFu); c.b=static_cast<s16>(ba>>16); c.a=static_cast<s16>(ba&0xFFFFu);
|
||||
GXSetTevColorS10((GXTevRegID)id, c);
|
||||
}
|
||||
extern "C" void GX__SetTevColorS10_80171e70(uint32_t id, uint32_t cp) {
|
||||
if (!TevRegOk(id)) return;
|
||||
const uint8_t* p=Memory::GetPointer(cp, 8); GXColorS10 c;
|
||||
c.r=(p[0]<<8)|p[1]; c.g=(p[2]<<8)|p[3]; c.b=(p[4]<<8)|p[5]; c.a=(p[6]<<8)|p[7];
|
||||
GXSetTevColorS10((GXTevRegID)id, c);
|
||||
GxThread::Post(&GX__SetTevColorS10_gx, id, Memory::Read32(cp), Memory::Read32(cp + 4));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171e70, GX__SetTevColorS10_80171e70, (uint32_t id, uint32_t cp), (id, cp));
|
||||
|
||||
void GX__SetTevKColor_gx(uint32_t id, uint32_t colorWord) {
|
||||
GXSetTevKColor((GXTevKColorID)id, DecodeGxColor(colorWord));
|
||||
}
|
||||
extern "C" void GX__SetTevKColor_80171ed4(uint32_t id, uint32_t cp) {
|
||||
if (!TevKColorOk(id)) return;
|
||||
const uint8_t* p=Memory::GetPointer(cp, 4); GXColor c; c.r=p[0]; c.g=p[1]; c.b=p[2]; c.a=p[3];
|
||||
GXSetTevKColor((GXTevKColorID)id, c);
|
||||
GxThread::Post(&GX__SetTevKColor_gx, id, Memory::Read32(cp));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171ed4, GX__SetTevKColor_80171ed4, (uint32_t id, uint32_t cp), (id, cp));
|
||||
|
||||
extern "C" void GX__SetTevKColorSel_80171f30(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKColorSel((GXTevStageID)s, (GXTevKColorSel)sel); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171f30, GX__SetTevKColorSel_80171f30, (uint32_t s, uint32_t sel), (s, sel));
|
||||
static void GX__SetTevKColorSel_80171f30_gx(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKColorSel((GXTevStageID)s, (GXTevKColorSel)sel); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171f30, GX__SetTevKColorSel_80171f30, (uint32_t s, uint32_t sel), (s, sel));
|
||||
|
||||
extern "C" void GX__SetTevKAlphaSel_80171f80(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKAlphaSel((GXTevStageID)s, (GXTevKAlphaSel)sel); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171f80, GX__SetTevKAlphaSel_80171f80, (uint32_t s, uint32_t sel), (s, sel));
|
||||
static void GX__SetTevKAlphaSel_80171f80_gx(uint32_t s, uint32_t sel) { if (!TevStageOk(s)) return; GXSetTevKAlphaSel((GXTevStageID)s, (GXTevKAlphaSel)sel); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171f80, GX__SetTevKAlphaSel_80171f80, (uint32_t s, uint32_t sel), (s, sel));
|
||||
|
||||
// ============================================================================
|
||||
// TEV Swap Tables
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetTevSwapModeTable_8017200c(uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a) {
|
||||
static void GX__SetTevSwapModeTable_8017200c_gx(uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a) {
|
||||
if (!TevSwapOk(id)) return;
|
||||
GXSetTevSwapModeTable((GXTevSwapSel)id, (GXTevColorChan)r, (GXTevColorChan)g, (GXTevColorChan)b, (GXTevColorChan)a);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017200c, GX__SetTevSwapModeTable_8017200c, (uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a), (id, r, g, b, a));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017200c, GX__SetTevSwapModeTable_8017200c, (uint32_t id, uint32_t r, uint32_t g, uint32_t b, uint32_t a), (id, r, g, b, a));
|
||||
|
||||
extern "C" void GX__SetTevSwapMode_80171fd0(uint32_t s, uint32_t rs, uint32_t ts) {
|
||||
static void GX__SetTevSwapMode_80171fd0_gx(uint32_t s, uint32_t rs, uint32_t ts) {
|
||||
if (!TevStageOk(s) || !TevSwapOk(rs) || !TevSwapOk(ts)) return;
|
||||
GXSetTevSwapMode((GXTevStageID)s, (GXTevSwapSel)rs, (GXTevSwapSel)ts);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171fd0, GX__SetTevSwapMode_80171fd0, (uint32_t s, uint32_t rs, uint32_t ts), (s, rs, ts));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171fd0, GX__SetTevSwapMode_80171fd0, (uint32_t s, uint32_t rs, uint32_t ts), (s, rs, ts));
|
||||
@@ -250,9 +250,8 @@ extern "C" void GX__InitTexObj_801707f8(uint32_t oa, uint32_t da, uint32_t w, ui
|
||||
oa, da, w, h, f, ws, wt, m, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u);
|
||||
}
|
||||
const uint32_t canonicalDataAddr = CanonicalizeGxMainRamAddress(da);
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex); GXTexObj* obj = CreateHostTexObj(oa); TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.userData=0; meta.needsUpload=true;
|
||||
GXInitTexObj(obj, GuestToHostPtr(da), (u16)w, (u16)h, (GXTexFmt)f, (GXTexWrapMode)ws, (GXTexWrapMode)wt, (GXBool)m); MarkHostTexObjConstructed(oa);
|
||||
// Also write to guest memory so reads work
|
||||
WriteGuestTexObj(oa, canonicalDataAddr, (u16)w, (u16)h, f, ws, wt, m != 0, false, 0);
|
||||
}
|
||||
@@ -291,9 +290,8 @@ extern "C" void GX__InitTexObjCI_80170a04(uint32_t oa, uint32_t da, uint32_t w,
|
||||
oa, da, w, h, f, ws, wt, m, tl, cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u);
|
||||
}
|
||||
const uint32_t canonicalDataAddr = CanonicalizeGxMainRamAddress(da);
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex); GXTexObj* obj = CreateHostTexObj(oa); TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.tlut=tl; meta.userData=0; meta.needsUpload=true;
|
||||
GXInitTexObjCI(obj, GuestToHostPtr(da), (u16)w, (u16)h, (GXCITexFmt)f, (GXTexWrapMode)ws, (GXTexWrapMode)wt, (GXBool)m, tl); MarkHostTexObjConstructed(oa);
|
||||
// Also write to guest memory so reads work
|
||||
WriteGuestTexObj(oa, canonicalDataAddr, (u16)w, (u16)h, f, ws, wt, m != 0, true, tl);
|
||||
// GXInitTexObjCI clears bit1 in the flags byte; keep guest memory consistent.
|
||||
@@ -306,10 +304,9 @@ extern "C" void GX__InitTexObjCI_80170a04(uint32_t oa, uint32_t da, uint32_t w,
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170a04, GX__InitTexObjCI_80170a04, (uint32_t oa, uint32_t da, uint32_t w, uint32_t h, uint32_t f, uint32_t ws, uint32_t wt, uint32_t m, uint32_t tl), (oa, da, w, h, f, ws, wt, m, tl));
|
||||
|
||||
extern "C" void GX__InitTexObjLOD_80170a4c(uint32_t oa, uint32_t mif, uint32_t maf, float mil, float mal, float lb, uint32_t bc, uint32_t el, uint32_t ma) {
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex); GXTexObj* obj = GetHostTexObj(oa); TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
float fmal = std::isfinite(mal) && mal >= 0.f ? mal : 0.f, fmil = std::isfinite(mil) && mil >= 0.f ? std::min(mil, fmal) : 0.f;
|
||||
meta.minFilter=mif; meta.magFilter=maf; meta.minLod=fmil; meta.maxLod=fmal; meta.lodBias=lb; meta.biasClamp=(bc!=0); meta.edgeLod=(el!=0); meta.maxAniso=ma;
|
||||
GXInitTexObjLOD(obj, (GXTexFilter)mif, (GXTexFilter)maf, fmil, fmal, lb, (GXBool)bc, (GXBool)el, (GXAnisotropy)ma);
|
||||
// Also write LOD info to guest memory
|
||||
WriteGuestTexObjLOD(oa, mif, maf, fmil, fmal, lb, bc != 0, el != 0, ma);
|
||||
}
|
||||
@@ -317,7 +314,6 @@ PPC_NATIVE_OVERRIDE_VOID(80170a4c, GX__InitTexObjLOD_80170a4c, (uint32_t oa, uin
|
||||
|
||||
extern "C" void GX__InitTexObjWrapMode_80170b50(uint32_t oa, uint32_t ws, uint32_t wt) {
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex);
|
||||
GXTexObj* obj = GetHostTexObj(oa);
|
||||
TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
meta.wrapS = ws;
|
||||
meta.wrapT = wt;
|
||||
@@ -327,16 +323,13 @@ extern "C" void GX__InitTexObjWrapMode_80170b50(uint32_t oa, uint32_t ws, uint32
|
||||
Memory::Write32(oa + 0x00, word0);
|
||||
} catch (...) {
|
||||
}
|
||||
GXInitTexObjWrapMode(obj, (GXTexWrapMode)ws, (GXTexWrapMode)wt);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170b50, GX__InitTexObjWrapMode_80170b50, (uint32_t oa, uint32_t ws, uint32_t wt), (oa, ws, wt));
|
||||
|
||||
extern "C" void GX__InitTexObjTlut_80170b64(uint32_t oa, uint32_t tl) {
|
||||
std::lock_guard<std::mutex> guard(g_texObjMutex);
|
||||
GXTexObj* obj = GetHostTexObj(oa);
|
||||
TexObjMeta& meta = GetTexObjMeta(oa);
|
||||
meta.tlut = tl;
|
||||
GXInitTexObjTlut(obj, tl);
|
||||
// Keep guest GXTexObj coherent for later GXLoadTexObj from memory.
|
||||
try {
|
||||
Memory::Write32(oa + 0x18, tl);
|
||||
@@ -365,11 +358,6 @@ extern "C" void GX__InitTexObjFilter_80170b6c(uint32_t oa, uint32_t minFilter, u
|
||||
Memory::Write32(oa + 0x00, word0);
|
||||
} catch (...) {}
|
||||
|
||||
// Update host texture object if it exists
|
||||
if (GXTexObj* obj = TryGetHostTexObj(oa)) {
|
||||
ApplyHostTexObjLod(obj, meta, (GXTexFilter)minFilter, (GXTexFilter)magFilter, meta.lodBias,
|
||||
(GXAnisotropy)meta.maxAniso);
|
||||
}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170b6c, GX__InitTexObjFilter_80170b6c, (uint32_t oa, uint32_t minFilter, uint32_t magFilter), (oa, minFilter, magFilter));
|
||||
|
||||
@@ -395,11 +383,6 @@ extern "C" void GX__InitTexObjLODBias_80170b94(uint32_t oa, float bias) {
|
||||
Memory::Write32(oa + 0x00, word0);
|
||||
} catch (...) {}
|
||||
|
||||
// Update host texture object if it exists
|
||||
if (GXTexObj* obj = TryGetHostTexObj(oa)) {
|
||||
ApplyHostTexObjLod(obj, meta, (GXTexFilter)meta.minFilter, (GXTexFilter)meta.magFilter,
|
||||
clampedBias, (GXAnisotropy)meta.maxAniso);
|
||||
}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170b94, GX__InitTexObjLODBias_80170b94, (uint32_t oa, float bias), (oa, bias));
|
||||
|
||||
@@ -411,11 +394,6 @@ extern "C" void GX__InitTexObjUserData_80170be8(uint32_t oa, uint32_t userData)
|
||||
// The RVL SDK stores this opaque guest pointer verbatim at GXTexObj + 0x10.
|
||||
WriteGuest32(oa + 0x10, userData, "GXInitTexObjUserData");
|
||||
|
||||
// Aurora keeps an expanded host-side object, so mirror the pointer when
|
||||
// that representation has already been constructed.
|
||||
if (GXTexObj* obj = TryGetHostTexObj(oa)) {
|
||||
GXInitTexObjUserData(obj, GuestToHostPtr(userData));
|
||||
}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170be8, GX__InitTexObjUserData_80170be8,
|
||||
(uint32_t oa, uint32_t userData), (oa, userData));
|
||||
@@ -450,7 +428,6 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
static uint32_t s_invalidMetaLogCount = 0;
|
||||
static uint32_t s_invalidTidLogCount = 0;
|
||||
static uint32_t s_invalidDimLogCount = 0;
|
||||
static uint32_t s_hostExceptionLogCount = 0;
|
||||
static uint32_t s_metaLookupLogCount = 0;
|
||||
static uint32_t s_unknownFormatLogCount = 0;
|
||||
static uint32_t s_invalidDataLogCount = 0;
|
||||
@@ -471,7 +448,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
RT_LOGF(RT_TAG_GX,
|
||||
"GXLoadTexObj rejected reason=no-metadata oa=0x%08X tid=%u\n", oa, tid);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -481,7 +458,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
"GXLoadTexObj rejected reason=invalid-meta oa=0x%08X tid=%u data=0x%08X %ux%u\n",
|
||||
oa, tid, meta.dataAddr, meta.width, meta.height);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
|
||||
return;
|
||||
}
|
||||
if (!IsReasonableTextureDimensions(meta.width, meta.height)) {
|
||||
@@ -490,7 +467,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
"GXLoadTexObj rejected reason=bad-dimensions oa=0x%08X tid=%u %ux%u fmt=0x%X\n",
|
||||
oa, tid, meta.width, meta.height, meta.format);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
|
||||
return;
|
||||
}
|
||||
// One gate, two reasons: every aurora-loadable format is also a known one,
|
||||
@@ -506,7 +483,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
known ? "unsupported-format" : "unknown-format", oa, tid, meta.format,
|
||||
meta.width, meta.height);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
|
||||
return;
|
||||
}
|
||||
const float maxMipLevel = static_cast<float>(ComputeMaxMipLevel(meta.width, meta.height));
|
||||
@@ -527,7 +504,7 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
meta.mipmap ? 1u : 0u, static_cast<uint32_t>(maxLod),
|
||||
cpu ? cpu->pc : 0u, cpu ? cpu->lr : 0u);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
GxThread::Post(&GxHostBindPlaceholder_gx, tid);
|
||||
return;
|
||||
}
|
||||
const GXTexWrapMode wrapS = SanitizeWrapMode(meta.wrapS);
|
||||
@@ -544,58 +521,140 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
meta.maxLod = maxLodSafe;
|
||||
meta.lodBias = lodBiasSafe;
|
||||
|
||||
// Check if host texture object exists, create if needed
|
||||
GXTexObj* obj = TryGetHostTexObj(oa);
|
||||
const bool isPalette = IsPaletteTexFormat(meta.format);
|
||||
bool needsInit = (obj == nullptr);
|
||||
const auto metaIt = g_TexObjMeta.find(oa);
|
||||
if (!needsInit && metaIt != g_TexObjMeta.end()) {
|
||||
const TexObjMeta& cached = metaIt->second;
|
||||
if (cached.width != meta.width || cached.height != meta.height || cached.format != meta.format ||
|
||||
cached.wrapS != meta.wrapS || cached.wrapT != meta.wrapT || cached.mipmap != meta.mipmap ||
|
||||
cached.minFilter != meta.minFilter || cached.magFilter != meta.magFilter ||
|
||||
cached.minLod != meta.minLod || cached.maxLod != meta.maxLod ||
|
||||
cached.lodBias != meta.lodBias || cached.biasClamp != meta.biasClamp ||
|
||||
cached.edgeLod != meta.edgeLod || cached.maxAniso != meta.maxAniso ||
|
||||
cached.tlut != meta.tlut || cached.userData != meta.userData) {
|
||||
needsInit = true;
|
||||
// The sanitized meta is what the GX thread builds the aurora object from
|
||||
// and what the next load compares against. The upload flag is consumed
|
||||
// here so a guest write re-uploads exactly once.
|
||||
GxTexObjLoad load{};
|
||||
load.objAddr = oa;
|
||||
load.tid = tid;
|
||||
load.upload = meta.needsUpload;
|
||||
meta.needsUpload = false;
|
||||
load.meta = meta;
|
||||
// Write through GetTexObjMeta so the DCStoreRange interval index is
|
||||
// told this entry's backing may have moved.
|
||||
GetTexObjMeta(oa) = meta;
|
||||
GxThread::Post(&GxHostLoadTexObj_gx, load);
|
||||
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) { Memory::Write32(gd + 0x5FCu, Memory::Read32(gd + 0x5FCu) | 1u); Memory::Write16(gd + 2, 0); } } catch (...) {}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170f2c, GX__LoadTexObj_80170f2c, (uint32_t oa, uint32_t tid), (oa, tid));
|
||||
|
||||
extern "C" void GX__LoadTexObjPreLoaded_80170dc8(uint32_t oa, uint32_t tid) { GX__LoadTexObj_80170f2c(oa, tid); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170dc8, GX__LoadTexObjPreLoaded_80170dc8, (uint32_t oa, uint32_t tid), (oa, tid));
|
||||
|
||||
// ============================================================================
|
||||
// GX-thread side: aurora texture and TLUT objects
|
||||
// ============================================================================
|
||||
// Keyed by guest object address like the meta table. Each load carries the
|
||||
// sanitized meta snapshot; an object is (re)built when it differs from what
|
||||
// the aurora object was last built from, exactly as the single-threaded
|
||||
// path compared the cached meta before this split.
|
||||
|
||||
namespace aurora::gfx {
|
||||
struct TextureRef;
|
||||
using TextureHandle = std::shared_ptr<TextureRef>;
|
||||
} // namespace aurora::gfx
|
||||
|
||||
namespace {
|
||||
struct HostGXTlutObj {
|
||||
alignas(GXTlutObj) std::byte publicStorage[sizeof(GXTlutObj)]{};
|
||||
aurora::gfx::TextureHandle ref;
|
||||
};
|
||||
|
||||
struct HleTlutObj {
|
||||
static constexpr size_t kTlutObjStorageSize =
|
||||
(sizeof(GXTlutObj) >= sizeof(HostGXTlutObj)) ? sizeof(GXTlutObj) : sizeof(HostGXTlutObj);
|
||||
static constexpr size_t kTlutObjStorageAlign =
|
||||
(alignof(GXTlutObj) >= alignof(HostGXTlutObj)) ? alignof(GXTlutObj) : alignof(HostGXTlutObj);
|
||||
using Storage = std::aligned_storage_t<kTlutObjStorageSize, kTlutObjStorageAlign>;
|
||||
Storage storage{};
|
||||
bool storageLive = false;
|
||||
bool constructed = false;
|
||||
HleTlutObj() = default;
|
||||
~HleTlutObj() { Destroy(); }
|
||||
HostGXTlutObj* HostObj() { return reinterpret_cast<HostGXTlutObj*>(&storage); }
|
||||
GXTlutObj* PublicPtr() { return reinterpret_cast<GXTlutObj*>(HostObj()->publicStorage); }
|
||||
void EnsureStorageLive() {
|
||||
if (!storageLive) {
|
||||
new (&storage) HostGXTlutObj();
|
||||
storageLive = true;
|
||||
}
|
||||
}
|
||||
bool tlutChanged = false;
|
||||
void Destroy() {
|
||||
if (storageLive) {
|
||||
GXDestroyTlutObj(PublicPtr());
|
||||
std::destroy_at(HostObj());
|
||||
std::memset(&storage, 0, sizeof(storage));
|
||||
storageLive = false;
|
||||
constructed = false;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
struct GxHostTexObjEntry {
|
||||
HleTexObj host;
|
||||
TexObjMeta cached;
|
||||
};
|
||||
struct GxHostTlutObjEntry {
|
||||
HleTlutObj host;
|
||||
TlutObjMeta cached;
|
||||
};
|
||||
std::unordered_map<uint32_t, GxHostTexObjEntry> g_gxHostTexObjs;
|
||||
std::map<uint32_t, GxHostTlutObjEntry> g_gxHostTlutObjs;
|
||||
|
||||
bool SameTexObjBuildMeta(const TexObjMeta& cached, const TexObjMeta& meta) {
|
||||
return cached.width == meta.width && cached.height == meta.height && cached.format == meta.format &&
|
||||
cached.wrapS == meta.wrapS && cached.wrapT == meta.wrapT && cached.mipmap == meta.mipmap &&
|
||||
cached.minFilter == meta.minFilter && cached.magFilter == meta.magFilter &&
|
||||
cached.minLod == meta.minLod && cached.maxLod == meta.maxLod &&
|
||||
cached.lodBias == meta.lodBias && cached.biasClamp == meta.biasClamp &&
|
||||
cached.edgeLod == meta.edgeLod && cached.maxAniso == meta.maxAniso &&
|
||||
cached.tlut == meta.tlut && cached.userData == meta.userData;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
void GxHostBindPlaceholder_gx(uint32_t tid) { BindUnloadableTexturePlaceholder(tid); }
|
||||
|
||||
void GxHostLoadTexObj_gx(GxTexObjLoad load) {
|
||||
static uint32_t s_hostExceptionLogCount = 0;
|
||||
const TexObjMeta& meta = load.meta;
|
||||
const uint32_t oa = load.objAddr;
|
||||
const uint32_t tid = load.tid;
|
||||
if (tid >= g_boundTexMaps.size()) {
|
||||
return;
|
||||
}
|
||||
const bool isPalette = IsPaletteTexFormat(meta.format);
|
||||
const uint8_t maxLod = (meta.maxLod > 0.0f) ? ((meta.maxLod > 255.0f) ? 255u : static_cast<uint8_t>(meta.maxLod)) : 0u;
|
||||
const uint32_t size = GXGetTexBufferSize(meta.width, meta.height, meta.format, (GXBool)meta.mipmap, maxLod);
|
||||
GxHostTexObjEntry& entry = g_gxHostTexObjs[oa];
|
||||
GXTexObj* obj = entry.host.constructed ? entry.host.PublicPtr() : nullptr;
|
||||
const bool needsInit = obj == nullptr || !SameTexObjBuildMeta(entry.cached, meta);
|
||||
bool textureDataUploaded = false;
|
||||
try {
|
||||
if (needsInit) {
|
||||
if (!obj) {
|
||||
obj = CreateHostTexObj(oa);
|
||||
}
|
||||
entry.host.Destroy();
|
||||
entry.host.EnsureStorageLive();
|
||||
obj = entry.host.PublicPtr();
|
||||
void* dp = GuestToHostPtr(meta.dataAddr, size);
|
||||
if (dp) {
|
||||
if (isPalette) {
|
||||
GXInitTexObjCI(obj, dp, meta.width, meta.height, (GXCITexFmt)meta.format,
|
||||
wrapS, wrapT,
|
||||
(GXTexWrapMode)meta.wrapS, (GXTexWrapMode)meta.wrapT,
|
||||
meta.mipmap ? GX_TRUE : GX_FALSE, meta.tlut);
|
||||
} else {
|
||||
GXInitTexObj(obj, dp, meta.width, meta.height, (GXTexFmt)meta.format,
|
||||
wrapS, wrapT,
|
||||
(GXTexWrapMode)meta.wrapS, (GXTexWrapMode)meta.wrapT,
|
||||
meta.mipmap ? GX_TRUE : GX_FALSE);
|
||||
}
|
||||
ApplyHostTexObjLod(obj, meta, minFilter, magFilter, meta.lodBias, maxAnisoSafe);
|
||||
ApplyHostTexObjLod(obj, meta, (GXTexFilter)meta.minFilter, (GXTexFilter)meta.magFilter,
|
||||
meta.lodBias, (GXAnisotropy)meta.maxAniso);
|
||||
GXInitTexObjUserData(obj, GuestToHostPtr(meta.userData));
|
||||
}
|
||||
MarkHostTexObjConstructed(oa);
|
||||
// Write through GetTexObjMeta so the DCStoreRange interval index is
|
||||
// told this entry's backing may have moved.
|
||||
GetTexObjMeta(oa) = meta;
|
||||
} else if (isPalette && metaIt != g_TexObjMeta.end() && metaIt->second.tlut != meta.tlut) {
|
||||
GXInitTexObjTlut(obj, meta.tlut);
|
||||
GetTexObjMeta(oa).tlut = meta.tlut;
|
||||
tlutChanged = true;
|
||||
entry.host.constructed = true;
|
||||
entry.cached = meta;
|
||||
}
|
||||
|
||||
const void* dp = GuestToHostPtr(meta.dataAddr, size);
|
||||
if (dp && meta.needsUpload) {
|
||||
if (dp && load.upload) {
|
||||
GXInitTexObjData(obj, dp);
|
||||
meta.needsUpload = false;
|
||||
textureDataUploaded = true;
|
||||
}
|
||||
// Everything the binding contract compares apart from objAddr is the
|
||||
@@ -607,10 +666,9 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
newBound.dataAddr = CanonicalizeGxMainRamAddress(meta.dataAddr);
|
||||
const BoundTexInfo& oldBound = g_boundTexMaps[tid];
|
||||
const bool canSkipHostLoad = GxTextureBindingContract::CanSkipHostLoad(
|
||||
oldBound, newBound, needsInit, tlutChanged, textureDataUploaded);
|
||||
|
||||
oldBound, newBound, needsInit, /*tlutChanged=*/false, textureDataUploaded);
|
||||
g_boundTexMaps[tid] = newBound;
|
||||
GetTexObjMeta(oa) = meta;
|
||||
entry.cached = meta;
|
||||
if (!canSkipHostLoad) {
|
||||
GXLoadTexObj(obj, (GXTexMapID)tid);
|
||||
}
|
||||
@@ -621,7 +679,6 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
ex.what(), oa, tid, meta.format, meta.width, meta.height, meta.dataAddr);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
return;
|
||||
} catch (...) {
|
||||
if (s_hostExceptionLogCount++ < 64) {
|
||||
RT_LOGF(RT_TAG_GX,
|
||||
@@ -629,14 +686,39 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
|
||||
oa, tid, meta.format, meta.width, meta.height, meta.dataAddr);
|
||||
}
|
||||
BindUnloadableTexturePlaceholder(tid);
|
||||
}
|
||||
}
|
||||
|
||||
void GxHostLoadTlut_gx(GxTlutLoad load) {
|
||||
GxHostTlutObjEntry& entry = g_gxHostTlutObjs[load.objAddr];
|
||||
if (!load.valid) {
|
||||
// GXInitTlutObj refused the descriptor (or the object was never
|
||||
// initialised): load a fresh empty object rather than read entries*2
|
||||
// bytes off an unvalidated pointer, as the soft-fail path always did.
|
||||
static uint32_t s_invalidLogCount = 0;
|
||||
if (s_invalidLogCount++ < 64) {
|
||||
RT_LOGF(RT_TAG_GX, "GXTex: invalid GXTlutObj @0x%08X\n", load.objAddr);
|
||||
}
|
||||
entry.host.Destroy();
|
||||
entry.host.EnsureStorageLive();
|
||||
entry.host.constructed = true;
|
||||
entry.cached = TlutObjMeta{};
|
||||
GXLoadTlut(entry.host.PublicPtr(), (GXTlut)load.tlut);
|
||||
return;
|
||||
}
|
||||
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) { Memory::Write32(gd + 0x5FCu, Memory::Read32(gd + 0x5FCu) | 1u); Memory::Write16(gd + 2, 0); } } catch (...) {}
|
||||
const bool metaChanged = entry.cached.dataAddr != load.meta.dataAddr ||
|
||||
entry.cached.format != load.meta.format ||
|
||||
entry.cached.entries != load.meta.entries;
|
||||
if (!entry.host.constructed || load.rebuild || metaChanged) {
|
||||
entry.host.Destroy();
|
||||
entry.host.EnsureStorageLive();
|
||||
GXInitTlutObj(entry.host.PublicPtr(), GuestToHostPtr(load.meta.dataAddr),
|
||||
(GXTlutFmt)load.meta.format, load.meta.entries);
|
||||
entry.host.constructed = true;
|
||||
entry.cached = load.meta;
|
||||
}
|
||||
GXLoadTlut(entry.host.PublicPtr(), (GXTlut)load.tlut);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170f2c, GX__LoadTexObj_80170f2c, (uint32_t oa, uint32_t tid), (oa, tid));
|
||||
|
||||
extern "C" void GX__LoadTexObjPreLoaded_80170dc8(uint32_t oa, uint32_t tid) { GX__LoadTexObj_80170f2c(oa, tid); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170dc8, GX__LoadTexObjPreLoaded_80170dc8, (uint32_t oa, uint32_t tid), (oa, tid));
|
||||
|
||||
// ============================================================================
|
||||
// Texture Object Getters
|
||||
@@ -676,33 +758,59 @@ PPC_NATIVE_OVERRIDE_VOID(80170cbc, GX__GetTexObjLODAll_80170cbc, (uint32_t oa, u
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__InitTlutObj_80170f80(uint32_t oa, uint32_t da, uint32_t f, uint32_t e) {
|
||||
std::lock_guard<std::mutex> guard(g_tlutObjMutex); GXTlutObj* obj = CreateHostTlutObj(oa); TlutObjMeta& meta = GetTlutObjMeta(oa);
|
||||
std::lock_guard<std::mutex> guard(g_tlutObjMutex); TlutObjMeta& meta = GetTlutObjMeta(oa);
|
||||
meta.dataAddr=CanonicalizeGxMainRamAddress(da); meta.format=f; meta.entries=(u16)e; meta.dirty=false;
|
||||
// Leave the object unconstructed on a bad descriptor. A later GXLoadTlut
|
||||
// then takes GetHostTlutObj's soft-fail path and gets a fresh empty object
|
||||
// instead of aurora reading entries*2 bytes off an unvalidated pointer.
|
||||
if (!ValidateTlutData(oa, meta)) return;
|
||||
GXInitTlutObj(obj, GuestToHostPtr(da), (GXTlutFmt)f, (u16)e); MarkHostTlutObjConstructed(oa);
|
||||
// The aurora object is built by the GX thread on the next GXLoadTlut from
|
||||
// this meta; a bad descriptor loads an empty object there instead of
|
||||
// aurora reading entries*2 bytes off an unvalidated pointer.
|
||||
(void)ValidateTlutData(oa, meta);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170f80, GX__InitTlutObj_80170f80, (uint32_t oa, uint32_t da, uint32_t f, uint32_t e), (oa, da, f, e));
|
||||
|
||||
extern "C" void GX__LoadTlut_80170fa8(uint32_t oa, uint32_t tl) { std::lock_guard<std::mutex> guard(g_tlutObjMutex); if (tl >= kMaxTluts) { RT_LOGF(RT_TAG_GX, "GXLoadTlut: invalid TLUT index %u (oa=0x%08X)\n", tl, oa); return; } auto metaIt = g_TlutObjMeta.find(oa); if (metaIt != g_TlutObjMeta.end() && metaIt->second.dirty) { if (ValidateTlutData(oa, metaIt->second)) { GXTlutObj* rebuild = CreateHostTlutObj(oa); GXInitTlutObj(rebuild, GuestToHostPtr(metaIt->second.dataAddr), (GXTlutFmt)metaIt->second.format, metaIt->second.entries); MarkHostTlutObjConstructed(oa); } metaIt->second.dirty = false; } GXTlutObj* obj = GetHostTlutObj(oa); GXLoadTlut(obj, (GXTlut)tl); try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {} }
|
||||
extern "C" void GX__LoadTlut_80170fa8(uint32_t oa, uint32_t tl) {
|
||||
std::lock_guard<std::mutex> guard(g_tlutObjMutex);
|
||||
if (tl >= kMaxTluts) {
|
||||
RT_LOGF(RT_TAG_GX, "GXLoadTlut: invalid TLUT index %u (oa=0x%08X)\n", tl, oa);
|
||||
return;
|
||||
}
|
||||
GxTlutLoad load{};
|
||||
load.objAddr = oa;
|
||||
load.tlut = tl;
|
||||
auto metaIt = g_TlutObjMeta.find(oa);
|
||||
if (metaIt != g_TlutObjMeta.end()) {
|
||||
TlutObjMeta& meta = metaIt->second;
|
||||
load.valid = ValidateTlutData(oa, meta);
|
||||
// A guest write over the palette (DCStoreRange) marks it dirty; the
|
||||
// rebuild travels with this load and the flag is consumed once.
|
||||
load.rebuild = meta.dirty && load.valid;
|
||||
meta.dirty = false;
|
||||
load.meta = meta;
|
||||
}
|
||||
GxThread::Post(&GxHostLoadTlut_gx, load);
|
||||
try { uint32_t gd = Memory::Read32(kGXDataPtrAddr); if (gd) Memory::Write16(gd + 2, 0); } catch (...) {}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80170fa8, GX__LoadTlut_80170fa8, (uint32_t oa, uint32_t tl), (oa, tl));
|
||||
|
||||
// ============================================================================
|
||||
// Texture Invalidation and Coordinate Control
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__InvalidateTexAll_80171110() {
|
||||
static void GX__InvalidateTexAll_80171110_gx() {
|
||||
// Real GX invalidates its internal texture cache here. Aurora forwards the
|
||||
// invalidate to the renderer, which keeps unchanged source uploads hot but
|
||||
// revalidates reused guest buffers before serving cached texture handles.
|
||||
GXInvalidateTexAll();
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171110, GX__InvalidateTexAll_80171110, (), ());
|
||||
GX_DEFERRED_OVERRIDE_VOID(80171110, GX__InvalidateTexAll_80171110, (), ());
|
||||
|
||||
extern "C" void GX__SetTexCoordScaleManually_80171180(uint32_t c, uint32_t en, uint32_t ss, uint32_t ts) { GXSetTexCoordScaleManually((GXTexCoordID)c, (GXBool)en, (u16)ss, (u16)ts); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ Memory::Write32(gd+0x5E4u, (Memory::Read32(gd+0x5E4u)&~(1u<<c))|((en&1u)<<c)); if(en){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFF0000u)|((ss-1)&0xFFFFu)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFF0000u)|((ts-1)&0xFFFFu)); Memory::Write16(gd+2, 0); } } }catch(...){} }
|
||||
static void GX__SetTexCoordScaleManually_gx(uint32_t c, uint32_t en, uint32_t ss, uint32_t ts) {
|
||||
GXSetTexCoordScaleManually((GXTexCoordID)c, (GXBool)en, (u16)ss, (u16)ts);
|
||||
}
|
||||
extern "C" void GX__SetTexCoordScaleManually_80171180(uint32_t c, uint32_t en, uint32_t ss, uint32_t ts) { GxThread::Post(&GX__SetTexCoordScaleManually_gx, c, en, ss, ts); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ Memory::Write32(gd+0x5E4u, (Memory::Read32(gd+0x5E4u)&~(1u<<c))|((en&1u)<<c)); if(en){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFF0000u)|((ss-1)&0xFFFFu)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFF0000u)|((ts-1)&0xFFFFu)); Memory::Write16(gd+2, 0); } } }catch(...){} }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80171180, GX__SetTexCoordScaleManually_80171180, (uint32_t c, uint32_t en, uint32_t ss, uint32_t ts), (c, en, ss, ts));
|
||||
|
||||
extern "C" void GX__SetTexCoordBias_801711fc(uint32_t c, uint32_t se, uint32_t te) { GXSetTexCoordBias((GXTexCoordID)c, (GXBool)se, (GXBool)te); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFEFFFFu)|((se&1u)<<16)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFEFFFFu)|((te&1u)<<16)); if(Memory::Read32(gd+0x5E4u)&(1u<<c)) Memory::Write16(gd+2, 0); } }catch(...){} }
|
||||
static void GX__SetTexCoordBias_gx(uint32_t c, uint32_t se, uint32_t te) {
|
||||
GXSetTexCoordBias((GXTexCoordID)c, (GXBool)se, (GXBool)te);
|
||||
}
|
||||
extern "C" void GX__SetTexCoordBias_801711fc(uint32_t c, uint32_t se, uint32_t te) { GxThread::Post(&GX__SetTexCoordBias_gx, c, se, te); try{ uint32_t gd=Memory::Read32(kGXDataPtrAddr); if(gd){ uint32_t sa=gd+0x108u+c*4u, ta=gd+0x128u+c*4u; Memory::Write32(sa, (Memory::Read32(sa)&0xFFFEFFFFu)|((se&1u)<<16)); Memory::Write32(ta, (Memory::Read32(ta)&0xFFFEFFFFu)|((te&1u)<<16)); if(Memory::Read32(gd+0x5E4u)&(1u<<c)) Memory::Write16(gd+2, 0); } }catch(...){} }
|
||||
PPC_NATIVE_OVERRIDE_VOID(801711fc, GX__SetTexCoordBias_801711fc, (uint32_t c, uint32_t se, uint32_t te), (c, se, te));
|
||||
@@ -0,0 +1,369 @@
|
||||
#include "gx_thread.h"
|
||||
#include "runtime_log.h"
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <condition_variable>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <exception>
|
||||
#include <mutex>
|
||||
#include <thread>
|
||||
#if defined(_WIN32)
|
||||
#include <windows.h>
|
||||
#else
|
||||
#include <pthread.h>
|
||||
#include <unistd.h>
|
||||
#if defined(__linux__)
|
||||
#include <sys/syscall.h>
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// The ring: one producer (the game thread) and one consumer. Records are
|
||||
// 16-byte headers followed by a payload, 16-byte aligned, always contiguous;
|
||||
// a header with a null invoke is a wrap marker that skips to the ring start.
|
||||
extern "C" void GxFifoConsumeBytes(const uint8_t* data, uint32_t sizeBytes);
|
||||
|
||||
namespace GxThread {
|
||||
namespace {
|
||||
|
||||
constexpr uint32_t kRingBytes = 16u << 20;
|
||||
constexpr uint32_t kRingMask = kRingBytes - 1u;
|
||||
constexpr uint32_t kHeaderBytes = 16;
|
||||
constexpr uint32_t kFifoChunkBytes = 8192;
|
||||
constexpr uint32_t kConsumerSpinIterations = 4000;
|
||||
|
||||
struct Header {
|
||||
uint64_t invoke;
|
||||
uint32_t payloadBytes;
|
||||
uint32_t stride;
|
||||
};
|
||||
|
||||
using Clock = std::chrono::steady_clock;
|
||||
|
||||
bool g_requested = false;
|
||||
bool g_enabled = false;
|
||||
bool g_started = false;
|
||||
std::thread g_thread;
|
||||
uint8_t* g_ring = nullptr;
|
||||
thread_local bool t_isGxThread = false;
|
||||
|
||||
alignas(64) std::atomic<uint64_t> g_head{0};
|
||||
alignas(64) std::atomic<uint64_t> g_tail{0};
|
||||
alignas(64) std::atomic<bool> g_stop{false};
|
||||
std::atomic<bool> g_consumerSleeping{false};
|
||||
std::atomic<bool> g_producerWaiting{false};
|
||||
std::atomic<uint64_t> g_fenceCompleted{0};
|
||||
std::atomic<uint32_t> g_nativeTid{0};
|
||||
std::mutex g_mutex;
|
||||
std::condition_variable g_cvData;
|
||||
std::condition_variable g_cvSpace;
|
||||
std::condition_variable g_cvFence;
|
||||
void (*g_waitCallback)() = nullptr;
|
||||
|
||||
// Producer-only state.
|
||||
uint64_t g_localTail = 0;
|
||||
uint64_t g_fenceRequested = 0;
|
||||
uint8_t g_pendingFifo[kFifoChunkBytes];
|
||||
uint32_t g_pendingFifoBytes = 0;
|
||||
|
||||
// Window statistics. Producer-side counters are plain; consumer-side ones are
|
||||
// relaxed atomics read by the producer when it formats the log line.
|
||||
uint64_t g_statRecords = 0;
|
||||
uint64_t g_statFifoRecords = 0;
|
||||
uint64_t g_statFifoBytes = 0;
|
||||
uint64_t g_statBytes = 0;
|
||||
uint64_t g_statDrains = 0;
|
||||
uint64_t g_statDrainWaitNs = 0;
|
||||
uint64_t g_statSpaceWaitNs = 0;
|
||||
uint64_t g_statQueuePeakBytes = 0;
|
||||
std::atomic<uint64_t> g_statBusyNs{0};
|
||||
std::atomic<uint64_t> g_statFaults{0};
|
||||
|
||||
uint64_t ElapsedNs(Clock::time_point since) {
|
||||
return static_cast<uint64_t>(std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - since).count());
|
||||
}
|
||||
|
||||
uint32_t Align16(uint32_t bytes) { return (bytes + 15u) & ~15u; }
|
||||
|
||||
void SetThreadName() {
|
||||
#if defined(_WIN32)
|
||||
using SetThreadDescriptionFn = HRESULT(WINAPI*)(HANDLE, PCWSTR);
|
||||
if (HMODULE kernel = ::GetModuleHandleW(L"kernel32.dll")) {
|
||||
if (auto fn = reinterpret_cast<SetThreadDescriptionFn>(::GetProcAddress(kernel, "SetThreadDescription"))) {
|
||||
fn(::GetCurrentThread(), L"MKW GX");
|
||||
}
|
||||
}
|
||||
g_nativeTid.store(static_cast<uint32_t>(::GetCurrentThreadId()), std::memory_order_release);
|
||||
#else
|
||||
#if defined(__APPLE__)
|
||||
pthread_setname_np("MKW GX");
|
||||
#else
|
||||
pthread_setname_np(pthread_self(), "MKW GX");
|
||||
#endif
|
||||
#if defined(__linux__)
|
||||
g_nativeTid.store(static_cast<uint32_t>(syscall(SYS_gettid)), std::memory_order_release);
|
||||
#else
|
||||
g_nativeTid.store(1u, std::memory_order_release);
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
// Waits on `cv` until pred() holds, running the wait callback between
|
||||
// bounded waits so the game thread's VI/alarm servicing never starves.
|
||||
template <typename Pred>
|
||||
void WaitWithCallback(std::condition_variable& cv, std::atomic<bool>& waitingFlag, Pred pred) {
|
||||
waitingFlag.store(true, std::memory_order_release);
|
||||
while (true) {
|
||||
{
|
||||
std::unique_lock<std::mutex> lock(g_mutex);
|
||||
if (cv.wait_for(lock, std::chrono::milliseconds(2), pred)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (g_waitCallback != nullptr) {
|
||||
g_waitCallback();
|
||||
}
|
||||
}
|
||||
waitingFlag.store(false, std::memory_order_release);
|
||||
}
|
||||
|
||||
void NotifyConsumer() {
|
||||
if (g_consumerSleeping.load(std::memory_order_acquire)) {
|
||||
std::lock_guard<std::mutex> lock(g_mutex);
|
||||
g_cvData.notify_one();
|
||||
}
|
||||
}
|
||||
|
||||
void PostRecordRaw(detail::Invoke invoke, const void* payload, uint32_t payloadBytes) {
|
||||
const uint32_t stride = Align16(kHeaderBytes + payloadBytes);
|
||||
uint32_t offset = static_cast<uint32_t>(g_localTail & kRingMask);
|
||||
const uint32_t wrapBytes = (offset + stride > kRingBytes) ? (kRingBytes - offset) : 0u;
|
||||
const uint64_t required = static_cast<uint64_t>(stride) + wrapBytes;
|
||||
const auto freeBytes = [&] { return kRingBytes - (g_localTail - g_head.load(std::memory_order_acquire)); };
|
||||
if (freeBytes() < required) {
|
||||
const auto started = Clock::now();
|
||||
WaitWithCallback(g_cvSpace, g_producerWaiting, [&] { return freeBytes() >= required; });
|
||||
g_statSpaceWaitNs += ElapsedNs(started);
|
||||
}
|
||||
if (wrapBytes != 0) {
|
||||
const Header wrap{0, 0, wrapBytes};
|
||||
std::memcpy(g_ring + offset, &wrap, sizeof(wrap));
|
||||
g_localTail += wrapBytes;
|
||||
offset = 0;
|
||||
}
|
||||
const Header header{reinterpret_cast<uint64_t>(invoke), payloadBytes, stride};
|
||||
std::memcpy(g_ring + offset, &header, sizeof(header));
|
||||
if (payloadBytes != 0) {
|
||||
std::memcpy(g_ring + offset + kHeaderBytes, payload, payloadBytes);
|
||||
}
|
||||
g_localTail += stride;
|
||||
g_tail.store(g_localTail, std::memory_order_release);
|
||||
++g_statRecords;
|
||||
g_statBytes += stride;
|
||||
const uint64_t queued = g_localTail - g_head.load(std::memory_order_relaxed);
|
||||
if (queued > g_statQueuePeakBytes) {
|
||||
g_statQueuePeakBytes = queued;
|
||||
}
|
||||
NotifyConsumer();
|
||||
}
|
||||
|
||||
void FifoInvoke(const uint8_t* payload, uint32_t payloadBytes) { GxFifoConsumeBytes(payload, payloadBytes); }
|
||||
|
||||
void FenceInvoke(const uint8_t* payload, uint32_t) {
|
||||
uint64_t sequence = 0;
|
||||
std::memcpy(&sequence, payload, sizeof(sequence));
|
||||
{
|
||||
std::lock_guard<std::mutex> lock(g_mutex);
|
||||
g_fenceCompleted.store(sequence, std::memory_order_release);
|
||||
}
|
||||
g_cvFence.notify_all();
|
||||
}
|
||||
|
||||
void ConsumerLoop() {
|
||||
t_isGxThread = true;
|
||||
SetThreadName();
|
||||
uint64_t head = g_head.load(std::memory_order_relaxed);
|
||||
uint32_t spins = 0;
|
||||
while (true) {
|
||||
const uint64_t tail = g_tail.load(std::memory_order_acquire);
|
||||
if (head == tail) {
|
||||
if (g_stop.load(std::memory_order_acquire)) {
|
||||
break;
|
||||
}
|
||||
if (spins < kConsumerSpinIterations) {
|
||||
++spins;
|
||||
std::this_thread::yield();
|
||||
continue;
|
||||
}
|
||||
g_consumerSleeping.store(true, std::memory_order_release);
|
||||
{
|
||||
std::unique_lock<std::mutex> lock(g_mutex);
|
||||
g_cvData.wait_for(lock, std::chrono::milliseconds(1), [&] {
|
||||
return g_tail.load(std::memory_order_acquire) != head || g_stop.load(std::memory_order_acquire);
|
||||
});
|
||||
}
|
||||
g_consumerSleeping.store(false, std::memory_order_release);
|
||||
continue;
|
||||
}
|
||||
spins = 0;
|
||||
Header header;
|
||||
std::memcpy(&header, g_ring + (head & kRingMask), sizeof(header));
|
||||
if (header.invoke != 0) {
|
||||
const auto started = Clock::now();
|
||||
const uint8_t* payload = g_ring + ((head + kHeaderBytes) & kRingMask);
|
||||
try {
|
||||
reinterpret_cast<detail::Invoke>(header.invoke)(payload, header.payloadBytes);
|
||||
} catch (const std::exception& ex) {
|
||||
const uint64_t faults = g_statFaults.fetch_add(1u, std::memory_order_relaxed) + 1u;
|
||||
if (faults <= 64) {
|
||||
RT_LOGF(RT_TAG_GX, "GX thread: command raised '%s' (n=%llu)\n", ex.what(),
|
||||
static_cast<unsigned long long>(faults));
|
||||
}
|
||||
} catch (...) {
|
||||
const uint64_t faults = g_statFaults.fetch_add(1u, std::memory_order_relaxed) + 1u;
|
||||
if (faults <= 64) {
|
||||
RT_LOGF(RT_TAG_GX, "GX thread: command raised an unknown exception (n=%llu)\n",
|
||||
static_cast<unsigned long long>(faults));
|
||||
}
|
||||
}
|
||||
g_statBusyNs.fetch_add(ElapsedNs(started), std::memory_order_relaxed);
|
||||
}
|
||||
head += header.stride;
|
||||
g_head.store(head, std::memory_order_release);
|
||||
if (g_producerWaiting.load(std::memory_order_acquire)) {
|
||||
std::lock_guard<std::mutex> lock(g_mutex);
|
||||
g_cvSpace.notify_one();
|
||||
}
|
||||
}
|
||||
t_isGxThread = false;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void Configure(bool enabled) { g_requested = enabled; }
|
||||
bool Enabled() noexcept { return g_enabled; }
|
||||
bool IsGxThread() noexcept { return t_isGxThread; }
|
||||
void SetWaitCallback(void (*callback)()) { g_waitCallback = callback; }
|
||||
uint32_t NativeThreadId() noexcept { return g_nativeTid.load(std::memory_order_acquire); }
|
||||
|
||||
void Start() {
|
||||
if (!g_requested || g_started) {
|
||||
return;
|
||||
}
|
||||
g_ring = static_cast<uint8_t*>(std::malloc(kRingBytes));
|
||||
if (g_ring == nullptr) {
|
||||
RT_LOGF(RT_TAG_GX, "GX thread: ring allocation failed; running the GX pipeline on the game thread\n");
|
||||
return;
|
||||
}
|
||||
g_started = true;
|
||||
g_enabled = true;
|
||||
g_stop.store(false, std::memory_order_release);
|
||||
g_thread = std::thread(ConsumerLoop);
|
||||
RT_LOGF(RT_TAG_GX, "GX thread started (%u MiB command ring)\n", kRingBytes >> 20);
|
||||
}
|
||||
|
||||
void Stop() {
|
||||
if (!g_started) {
|
||||
return;
|
||||
}
|
||||
Drain();
|
||||
{
|
||||
std::lock_guard<std::mutex> lock(g_mutex);
|
||||
g_stop.store(true, std::memory_order_release);
|
||||
}
|
||||
g_cvData.notify_all();
|
||||
g_thread.join();
|
||||
g_started = false;
|
||||
g_enabled = false;
|
||||
}
|
||||
|
||||
void FlushFifo() {
|
||||
if (g_pendingFifoBytes == 0) {
|
||||
return;
|
||||
}
|
||||
const uint32_t bytes = g_pendingFifoBytes;
|
||||
g_pendingFifoBytes = 0;
|
||||
++g_statFifoRecords;
|
||||
g_statFifoBytes += bytes;
|
||||
PostRecordRaw(&FifoInvoke, g_pendingFifo, bytes);
|
||||
}
|
||||
|
||||
void Drain() {
|
||||
if (!g_enabled || !g_started || t_isGxThread) {
|
||||
return;
|
||||
}
|
||||
FlushFifo();
|
||||
const uint64_t sequence = ++g_fenceRequested;
|
||||
PostRecordRaw(&FenceInvoke, &sequence, sizeof(sequence));
|
||||
++g_statDrains;
|
||||
if (g_fenceCompleted.load(std::memory_order_acquire) >= sequence) {
|
||||
return;
|
||||
}
|
||||
const auto started = Clock::now();
|
||||
WaitWithCallback(g_cvFence, g_producerWaiting,
|
||||
[&] { return g_fenceCompleted.load(std::memory_order_acquire) >= sequence; });
|
||||
g_statDrainWaitNs += ElapsedNs(started);
|
||||
}
|
||||
|
||||
void PostFifoWord(uint32_t value, uint32_t sizeBytes) {
|
||||
if (sizeBytes == 0 || sizeBytes > 4) {
|
||||
sizeBytes = 4;
|
||||
}
|
||||
if (g_pendingFifoBytes + sizeBytes > kFifoChunkBytes) {
|
||||
FlushFifo();
|
||||
}
|
||||
uint8_t* out = g_pendingFifo + g_pendingFifoBytes;
|
||||
for (uint32_t i = 0; i < sizeBytes; ++i) {
|
||||
out[i] = static_cast<uint8_t>(value >> (8u * (sizeBytes - 1u - i)));
|
||||
}
|
||||
g_pendingFifoBytes += sizeBytes;
|
||||
}
|
||||
|
||||
void PostFifoBytes(const uint8_t* data, uint32_t sizeBytes) {
|
||||
if (data == nullptr || sizeBytes == 0) {
|
||||
return;
|
||||
}
|
||||
if (g_pendingFifoBytes + sizeBytes > kFifoChunkBytes) {
|
||||
FlushFifo();
|
||||
}
|
||||
if (sizeBytes > kFifoChunkBytes) {
|
||||
++g_statFifoRecords;
|
||||
g_statFifoBytes += sizeBytes;
|
||||
PostRecordRaw(&FifoInvoke, data, sizeBytes);
|
||||
return;
|
||||
}
|
||||
std::memcpy(g_pendingFifo + g_pendingFifoBytes, data, sizeBytes);
|
||||
g_pendingFifoBytes += sizeBytes;
|
||||
}
|
||||
|
||||
std::string FormatStatsAndReset(double windowSeconds, uint32_t frames) {
|
||||
const double perFrame = frames != 0 ? 1.0 / static_cast<double>(frames) : 0.0;
|
||||
const double busyNs = static_cast<double>(g_statBusyNs.exchange(0, std::memory_order_relaxed));
|
||||
const double busyPercent = windowSeconds > 0.0 ? busyNs / (windowSeconds * 1e9) * 100.0 : 0.0;
|
||||
char buffer[512];
|
||||
std::snprintf(buffer, sizeof(buffer),
|
||||
"GX thread: %.0f records/frame (%.1f KiB, %.0f FIFO chunks with %.1f KiB), queue peak %.1f KiB; "
|
||||
"game thread waited %.2f ms/frame for ring space and %.2f ms/frame in %.1f drains/frame; "
|
||||
"GX thread busy %.1f%%; faults %llu",
|
||||
static_cast<double>(g_statRecords) * perFrame,
|
||||
static_cast<double>(g_statBytes) * perFrame / 1024.0,
|
||||
static_cast<double>(g_statFifoRecords) * perFrame,
|
||||
static_cast<double>(g_statFifoBytes) * perFrame / 1024.0,
|
||||
static_cast<double>(g_statQueuePeakBytes) / 1024.0,
|
||||
static_cast<double>(g_statSpaceWaitNs) * perFrame / 1e6,
|
||||
static_cast<double>(g_statDrainWaitNs) * perFrame / 1e6,
|
||||
static_cast<double>(g_statDrains) * perFrame, busyPercent,
|
||||
static_cast<unsigned long long>(g_statFaults.load(std::memory_order_relaxed)));
|
||||
g_statRecords = g_statFifoRecords = g_statFifoBytes = g_statBytes = 0;
|
||||
g_statDrains = g_statDrainWaitNs = g_statSpaceWaitNs = g_statQueuePeakBytes = 0;
|
||||
return buffer;
|
||||
}
|
||||
|
||||
namespace detail {
|
||||
void PostRecord(Invoke invoke, const void* payload, uint32_t payloadBytes) {
|
||||
FlushFifo();
|
||||
PostRecordRaw(invoke, payload, payloadBytes);
|
||||
}
|
||||
} // namespace detail
|
||||
|
||||
} // namespace GxThread
|
||||
@@ -29,6 +29,30 @@ namespace {
|
||||
std::memcpy(g_projectionVector, projV, sizeof(g_projectionVector));
|
||||
}
|
||||
|
||||
// The SDK copies matrices into the FIFO at call time, so immediate loads
|
||||
// are snapshotted on the game thread; indexed loads read guest memory when
|
||||
// the GX thread reaches them, as the GP does.
|
||||
struct GxMtxSnapshot {
|
||||
float m[16];
|
||||
};
|
||||
|
||||
static void GX__SetViewportJitter_gx(float l, float t, float w, float h, float nz, float fz, uint32_t f) {
|
||||
GXSetViewportJitter(l, t, w, h, nz, fz, f);
|
||||
}
|
||||
static void GX__SetViewport_gx(float l, float t, float w, float h, float nz, float fz) {
|
||||
GXSetViewport(l, t, w, h, nz, fz);
|
||||
}
|
||||
static void GX__SetZScaleOffset_gx(float s, float o) { GXSetZScaleOffset(s, o); }
|
||||
static void GX__SetScissorBoxOffset_gx(int32_t xo, int32_t yo) { GXSetScissorBoxOffset(xo, yo); }
|
||||
static void GX__SetScissor_gx(uint32_t l, uint32_t t, uint32_t w, uint32_t h) { GXSetScissor(l, t, w, h); }
|
||||
static void GX__SetProjection_gx(GxMtxSnapshot proj, uint32_t pt) {
|
||||
GXSetProjection(proj.m, (GXProjectionType)pt);
|
||||
}
|
||||
static void GX__LoadPosMtxImm_gx(GxMtxSnapshot mtx, uint32_t id) { GXLoadPosMtxImm((float(*)[4])mtx.m, id); }
|
||||
static void GX__LoadNrmMtxImm_gx(GxMtxSnapshot mtx, uint32_t id) { GXLoadNrmMtxImm((float(*)[4])mtx.m, id); }
|
||||
static void GX__LoadTexMtxImm_gx(GxMtxSnapshot mtx, uint32_t id, uint32_t t) {
|
||||
GXLoadTexMtxImm(mtx.m, id, (GXTexMtxType)t);
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -40,13 +64,13 @@ namespace {
|
||||
// transform double-transformed menu viewports and broke 640x480 overscan, so it was removed.
|
||||
extern "C" void GX__SetViewportJitter_80173378(float l, float t, float w, float h, float nz, float fz, uint32_t f) {
|
||||
g_viewportState[0]=l; g_viewportState[1]=t; g_viewportState[2]=w; g_viewportState[3]=h; g_viewportState[4]=nz; g_viewportState[5]=fz;
|
||||
GXSetViewportJitter(l, t, w, h, nz, fz, f);
|
||||
GxThread::Post(&GX__SetViewportJitter_gx, l, t, w, h, nz, fz, f);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173378, GX__SetViewportJitter_80173378, (float l, float t, float w, float h, float nz, float fz, uint32_t f), (l, t, w, h, nz, fz, f));
|
||||
|
||||
extern "C" void GX__SetViewport_801733b4(float l, float t, float w, float h, float nz, float fz) {
|
||||
g_viewportState[0]=l; g_viewportState[1]=t; g_viewportState[2]=w; g_viewportState[3]=h; g_viewportState[4]=nz; g_viewportState[5]=fz;
|
||||
GXSetViewport(l, t, w, h, nz, fz);
|
||||
GxThread::Post(&GX__SetViewport_gx, l, t, w, h, nz, fz);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801733b4, GX__SetViewport_801733b4, (float l, float t, float w, float h, float nz, float fz), (l, t, w, h, nz, fz));
|
||||
|
||||
@@ -66,7 +90,7 @@ extern "C" void GX__GetViewportv_801733e0(uint32_t oa) {
|
||||
PPC_NATIVE_OVERRIDE_VOID(801733e0, GX__GetViewportv_801733e0, (uint32_t oa), (oa));
|
||||
|
||||
extern "C" void GX__SetZScaleOffset_80173400(float s, float o) {
|
||||
GXSetZScaleOffset(s, o);
|
||||
GxThread::Post(&GX__SetZScaleOffset_gx, s, o);
|
||||
try {
|
||||
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
|
||||
if (gd) {
|
||||
@@ -80,7 +104,7 @@ extern "C" void GX__SetZScaleOffset_80173400(float s, float o) {
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173400, GX__SetZScaleOffset_80173400, (float s, float o), (s, o));
|
||||
|
||||
extern "C" void GX__SetScissorBoxOffset_801734e0(int32_t xo, int32_t yo) {
|
||||
GXSetScissorBoxOffset(xo, yo);
|
||||
GxThread::Post(&GX__SetScissorBoxOffset_gx, xo, yo);
|
||||
try {
|
||||
const uint32_t gd = Memory::Read32(kGXDataPtrAddr);
|
||||
if (gd) Memory::Write16(gd + 2, 0);
|
||||
@@ -104,7 +128,7 @@ extern "C" void GX__SetScissor_80173430(uint32_t l, uint32_t t, uint32_t w, uint
|
||||
// No viewport replay here: aurora recomputes viewport and scissor together
|
||||
// on every scissor change (set_logical_scissor -> apply_logical_render_state),
|
||||
// so re-issuing the current viewport would be duplicate work.
|
||||
GXSetScissor(l, t, w, h);
|
||||
GxThread::Post(&GX__SetScissor_gx, l, t, w, h);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173430, GX__SetScissor_80173430, (uint32_t l, uint32_t t, uint32_t w, uint32_t h), (l, t, w, h));
|
||||
|
||||
@@ -113,10 +137,10 @@ PPC_NATIVE_OVERRIDE_VOID(80173430, GX__SetScissor_80173430, (uint32_t l, uint32_
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetProjection_8017301c(uint32_t ma, uint32_t pt) {
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 64); float m[16];
|
||||
SwapBeF32ArrayToHost(raw, m, 16);
|
||||
GXSetProjection(m, (GXProjectionType)pt);
|
||||
UpdateProjectionVectorFromMatrix(m, (GXProjectionType)pt);
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 64); GxMtxSnapshot proj{};
|
||||
SwapBeF32ArrayToHost(raw, proj.m, 16);
|
||||
UpdateProjectionVectorFromMatrix(proj.m, (GXProjectionType)pt);
|
||||
GxThread::Post(&GX__SetProjection_gx, proj, pt);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017301c, GX__SetProjection_8017301c, (uint32_t ma, uint32_t pt), (ma, pt));
|
||||
|
||||
@@ -124,10 +148,10 @@ extern "C" void GX__SetProjectionv_80173080(uint32_t pa) {
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(pa, 28); float v[7];
|
||||
SwapBeF32ArrayToHost(raw, v, 7);
|
||||
GXProjectionType pt=(v[0]!=0.f)?GX_ORTHOGRAPHIC:GX_PERSPECTIVE;
|
||||
float m[16]={0.f}; m[0]=v[1]; m[5]=v[3]; m[10]=v[5]; m[11]=v[6];
|
||||
GxMtxSnapshot proj{}; float* m = proj.m; m[0]=v[1]; m[5]=v[3]; m[10]=v[5]; m[11]=v[6];
|
||||
if(pt==GX_PERSPECTIVE){ m[2]=v[2]; m[6]=v[4]; m[14]=-1.f; } else { m[3]=v[2]; m[7]=v[4]; m[15]=1.f; }
|
||||
GXSetProjection(m, pt);
|
||||
UpdateProjectionVectorFromProjV(v);
|
||||
GxThread::Post(&GX__SetProjection_gx, proj, static_cast<uint32_t>(pt));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173080, GX__SetProjectionv_80173080, (uint32_t pa), (pa));
|
||||
|
||||
@@ -144,27 +168,28 @@ PPC_NATIVE_OVERRIDE_VOID(801730cc, GX__GetProjectionv_801730cc, (uint32_t pa), (
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__LoadPosMtxImm_8017310c(uint32_t ma, uint32_t id) {
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma); float m[12];
|
||||
SwapBeF32ArrayToHost(raw, m, 12);
|
||||
GXLoadPosMtxImm((float(*)[4])m, id);
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma); GxMtxSnapshot mtx{};
|
||||
SwapBeF32ArrayToHost(raw, mtx.m, 12);
|
||||
GxThread::Post(&GX__LoadPosMtxImm_gx, mtx, id);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017310c, GX__LoadPosMtxImm_8017310c, (uint32_t ma, uint32_t id), (ma, id));
|
||||
|
||||
extern "C" void GX__LoadPosMtxIndx_8017315c(uint32_t mi, uint32_t id) {
|
||||
static void GX__LoadPosMtxIndx_8017315c_gx(uint32_t mi, uint32_t id) {
|
||||
const auto& arr=g_hleGxState.vtxArray[GX_POS_MTX_ARRAY];
|
||||
if(arr.base==0||arr.stride==0) return;
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(arr.base+mi*arr.stride, 48);
|
||||
if(raw){ float m[12]; SwapBeF32ArrayToHost(raw,m,12); GXLoadPosMtxImm((float(*)[4])m,id); }
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8017315c, GX__LoadPosMtxIndx_8017315c, (uint32_t mi, uint32_t id), (mi, id));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8017315c, GX__LoadPosMtxIndx_8017315c, (uint32_t mi, uint32_t id), (mi, id));
|
||||
|
||||
extern "C" void GX__LoadNrmMtxImm_80173188(uint32_t ma, uint32_t id) {
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 48); float m[12];
|
||||
SwapBeF32ArrayToHost(raw, m, 12); GXLoadNrmMtxImm((float(*)[4])m, id);
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma, 48); GxMtxSnapshot mtx{};
|
||||
SwapBeF32ArrayToHost(raw, mtx.m, 12);
|
||||
GxThread::Post(&GX__LoadNrmMtxImm_gx, mtx, id);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173188, GX__LoadNrmMtxImm_80173188, (uint32_t ma, uint32_t id), (ma, id));
|
||||
|
||||
extern "C" void GX__LoadNrmMtxIndx3x3_801731e0(uint32_t mi, uint32_t id) {
|
||||
static void GX__LoadNrmMtxIndx3x3_801731e0_gx(uint32_t mi, uint32_t id) {
|
||||
const auto& arr=g_hleGxState.vtxArray[GX_NRM_MTX_ARRAY];
|
||||
if(arr.base==0||arr.stride==0) return;
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(arr.base+mi*arr.stride, 36);
|
||||
@@ -180,14 +205,15 @@ extern "C" void GX__LoadNrmMtxIndx3x3_801731e0(uint32_t mi, uint32_t id) {
|
||||
GXLoadNrmMtxImm((float(*)[4])m, id);
|
||||
}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(801731e0, GX__LoadNrmMtxIndx3x3_801731e0, (uint32_t mi, uint32_t id), (mi, id));
|
||||
GX_DEFERRED_OVERRIDE_VOID(801731e0, GX__LoadNrmMtxIndx3x3_801731e0, (uint32_t mi, uint32_t id), (mi, id));
|
||||
|
||||
extern "C" void GX__LoadTexMtxImm_80173234(uint32_t ma, uint32_t id, uint32_t t) {
|
||||
size_t c=(t==(uint32_t)GX_MTX3x4)?12:8;
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma,c*4); float l[12]={};
|
||||
SwapBeF32ArrayToHost(raw,l,c); GXLoadTexMtxImm(l, id, (GXTexMtxType)t);
|
||||
const uint32_t* raw=(const uint32_t*)GuestToHostPtr(ma,c*4); GxMtxSnapshot mtx{};
|
||||
SwapBeF32ArrayToHost(raw, mtx.m, c);
|
||||
GxThread::Post(&GX__LoadTexMtxImm_gx, mtx, id, t);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173234, GX__LoadTexMtxImm_80173234, (uint32_t ma, uint32_t id, uint32_t t), (ma, id, t));
|
||||
|
||||
extern "C" void GX__SetCurrentMtx_80173214(uint32_t id) { GXSetCurrentMtx(id); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80173214, GX__SetCurrentMtx_80173214, (uint32_t id), (id));
|
||||
static void GX__SetCurrentMtx_80173214_gx(uint32_t id) { GXSetCurrentMtx(id); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(80173214, GX__SetCurrentMtx_80173214, (uint32_t id), (id));
|
||||
@@ -8,6 +8,10 @@
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
|
||||
// Game-thread mirror for the GXGet* overrides; the parser state in
|
||||
// g_hleGxState belongs to the GX thread.
|
||||
GxGameSideVertexState g_gxGameVertexState;
|
||||
|
||||
namespace {
|
||||
uint32_t CanonicalVtxAttr(uint32_t attr) {
|
||||
return attr == GX_VA_NBT ? GX_VA_NRM : attr;
|
||||
@@ -31,7 +35,7 @@ void ApplyAuroraVtxStateForBegin(GXVtxFmt fmt) {
|
||||
// Vertex Descriptor
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__ClearVtxDesc_8016dc34() {
|
||||
static void GX__ClearVtxDesc_8016dc34_gx() {
|
||||
bool changed = false;
|
||||
for(int i=0; i<26; ++i){
|
||||
changed |= g_hleGxState.vtxDesc[i] != GX_NONE;
|
||||
@@ -41,9 +45,15 @@ extern "C" void GX__ClearVtxDesc_8016dc34() {
|
||||
// GXClearVtxDesc resets descriptors only; array base/stride state persists.
|
||||
GXClearVtxDesc();
|
||||
}
|
||||
extern "C" void GX__ClearVtxDesc_8016dc34() {
|
||||
for (int i = 0; i < 26; ++i) {
|
||||
g_gxGameVertexState.vtxDesc[i] = GX_NONE;
|
||||
}
|
||||
GxThread::Post(&GX__ClearVtxDesc_8016dc34_gx);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016dc34, GX__ClearVtxDesc_8016dc34, (), ());
|
||||
|
||||
extern "C" void GX__SetVtxDesc_8016d3a4(uint32_t a, uint32_t t) {
|
||||
static void GX__SetVtxDesc_8016d3a4_gx(uint32_t a, uint32_t t) {
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if(attr>=26||attr==GX_VA_NULL) return;
|
||||
const GXAttrType oldType = g_hleGxState.vtxDesc[attr];
|
||||
@@ -55,6 +65,16 @@ extern "C" void GX__SetVtxDesc_8016d3a4(uint32_t a, uint32_t t) {
|
||||
if(IsMatrixIndexAttr((GXAttr)attr)) return;
|
||||
GXSetVtxDesc((GXAttr)a, (t==GX_INDEX8||t==GX_INDEX16)?GX_DIRECT:(GXAttrType)t);
|
||||
}
|
||||
extern "C" void GX__SetVtxDesc_8016d3a4(uint32_t a, uint32_t t) {
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if (attr < 26 && attr != GX_VA_NULL) {
|
||||
g_gxGameVertexState.vtxDesc[attr] = (GXAttrType)t;
|
||||
if (a == GX_VA_NBT) {
|
||||
g_gxGameVertexState.vtxDesc[GX_VA_NBT] = GX_NONE;
|
||||
}
|
||||
}
|
||||
GxThread::Post(&GX__SetVtxDesc_8016d3a4_gx, a, t);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016d3a4, GX__SetVtxDesc_8016d3a4, (uint32_t a, uint32_t t), (a, t));
|
||||
|
||||
extern "C" void GX__SetVtxDescv_8016d608(uint32_t la) {
|
||||
@@ -72,7 +92,7 @@ extern "C" void GX__GetVtxDesc_8016d9f0(uint32_t a, uint32_t tp) {
|
||||
GXAttrType type = GX_NONE;
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if (attr < GX_VA_MAX_ATTR && attr != GX_VA_NULL) {
|
||||
type = g_hleGxState.vtxDesc[attr];
|
||||
type = g_gxGameVertexState.vtxDesc[attr];
|
||||
}
|
||||
if (tp) {
|
||||
Memory::Write32(tp, static_cast<uint32_t>(type));
|
||||
@@ -88,11 +108,11 @@ extern "C" void GX__GetVtxDescv_8016dba4(uint32_t la) {
|
||||
uint32_t p = la;
|
||||
for (uint32_t a = GX_VA_PNMTXIDX; a <= GX_VA_TEX7; ++a) {
|
||||
Memory::Write32(p, a);
|
||||
Memory::Write32(p + 4, static_cast<uint32_t>(g_hleGxState.vtxDesc[a]));
|
||||
Memory::Write32(p + 4, static_cast<uint32_t>(g_gxGameVertexState.vtxDesc[a]));
|
||||
p += 8;
|
||||
}
|
||||
Memory::Write32(p, GX_VA_NBT);
|
||||
Memory::Write32(p + 4, static_cast<uint32_t>(g_hleGxState.vtxDesc[GX_VA_NRM]));
|
||||
Memory::Write32(p + 4, static_cast<uint32_t>(g_gxGameVertexState.vtxDesc[GX_VA_NRM]));
|
||||
p += 8;
|
||||
Memory::Write32(p, GX_VA_NULL);
|
||||
}
|
||||
@@ -102,7 +122,7 @@ PPC_NATIVE_OVERRIDE_VOID(8016dba4, GX__GetVtxDescv_8016dba4, (uint32_t la), (la)
|
||||
// Vertex Attribute Format
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetVtxAttrFmt_8016dc68(uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr) {
|
||||
static void GX__SetVtxAttrFmt_8016dc68_gx(uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr) {
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if(vf<8&&attr<26){
|
||||
const VtxAttrFmt oldFmt = g_hleGxState.vtxAttrFmt[vf][attr];
|
||||
@@ -122,6 +142,19 @@ extern "C" void GX__SetVtxAttrFmt_8016dc68(uint32_t vf, uint32_t a, uint32_t c,
|
||||
}
|
||||
GXSetVtxAttrFmt((GXVtxFmt)vf, (GXAttr)a, (GXCompCnt)c, (GXCompType)t, (u8)fr);
|
||||
}
|
||||
extern "C" void GX__SetVtxAttrFmt_8016dc68(uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr) {
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if (vf < 8 && attr < 26) {
|
||||
VtxAttrFmt& fmt = g_gxGameVertexState.vtxAttrFmt[vf][attr];
|
||||
fmt.cnt = (GXCompCnt)c;
|
||||
fmt.type = (GXCompType)t;
|
||||
fmt.frac = (u8)fr;
|
||||
if (a == GX_VA_NBT) {
|
||||
g_gxGameVertexState.vtxAttrFmt[vf][GX_VA_NBT] = {};
|
||||
}
|
||||
}
|
||||
GxThread::Post(&GX__SetVtxAttrFmt_8016dc68_gx, vf, a, c, t, fr);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016dc68, GX__SetVtxAttrFmt_8016dc68, (uint32_t vf, uint32_t a, uint32_t c, uint32_t t, uint32_t fr), (vf, a, c, t, fr));
|
||||
|
||||
extern "C" void GX__SetVtxAttrFmtv_8016de08(uint32_t vf, uint32_t la) {
|
||||
@@ -139,7 +172,7 @@ extern "C" void GX__GetVtxAttrFmt_8016e04c(uint32_t vf, uint32_t a, uint32_t cp,
|
||||
VtxAttrFmt fmt{};
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if (vf < 8 && attr < GX_VA_MAX_ATTR && attr != GX_VA_NULL) {
|
||||
fmt = g_hleGxState.vtxAttrFmt[vf][attr];
|
||||
fmt = g_gxGameVertexState.vtxAttrFmt[vf][attr];
|
||||
}
|
||||
if (cp) {
|
||||
Memory::Write32(cp, static_cast<uint32_t>(fmt.cnt));
|
||||
@@ -162,7 +195,7 @@ extern "C" void GX__GetVtxAttrFmtv_8016e2b8(uint32_t vf, uint32_t la) {
|
||||
for (uint32_t a = GX_VA_POS; a <= GX_VA_TEX7; ++a) {
|
||||
VtxAttrFmt fmt{};
|
||||
if (vf < 8) {
|
||||
fmt = g_hleGxState.vtxAttrFmt[vf][a];
|
||||
fmt = g_gxGameVertexState.vtxAttrFmt[vf][a];
|
||||
}
|
||||
Memory::Write32(p, a);
|
||||
Memory::Write32(p + 4, static_cast<uint32_t>(fmt.cnt));
|
||||
@@ -178,11 +211,11 @@ PPC_NATIVE_OVERRIDE_VOID(8016e2b8, GX__GetVtxAttrFmtv_8016e2b8, (uint32_t vf, ui
|
||||
// Vertex Arrays
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetArray_8016e32c(uint32_t a, uint32_t ba, uint32_t str) {
|
||||
static void GX__SetArray_8016e32c_gx(uint32_t a, uint32_t ba, uint32_t str) {
|
||||
const uint32_t attr = CanonicalVtxAttr(a);
|
||||
if(attr<26){ g_hleGxState.vtxArray[attr].base=ba; g_hleGxState.vtxArray[attr].stride=str; }
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016e32c, GX__SetArray_8016e32c, (uint32_t a, uint32_t ba, uint32_t str), (a, ba, str));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016e32c, GX__SetArray_8016e32c, (uint32_t a, uint32_t ba, uint32_t str), (a, ba, str));
|
||||
|
||||
// Switch-artifact entry point for the same SDK function; forwards rather than
|
||||
// repeating the body.
|
||||
@@ -193,12 +226,12 @@ PPC_NATIVE_OVERRIDE_VOID(8016e1c4, GX__SetArray_8016e1c4, (uint32_t a, uint32_t
|
||||
// Texture Coordinate Generation
|
||||
// ============================================================================
|
||||
|
||||
extern "C" void GX__SetNumTexGens_8016e5a4(uint32_t n) {
|
||||
static void GX__SetNumTexGens_8016e5a4_gx(uint32_t n) {
|
||||
GXSetNumTexGens((u8)n);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016e5a4, GX__SetNumTexGens_8016e5a4, (uint32_t n), (n));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016e5a4, GX__SetNumTexGens_8016e5a4, (uint32_t n), (n));
|
||||
|
||||
extern "C" void GX__SetTexCoordGen2_8016e37c(uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm) {
|
||||
static void GX__SetTexCoordGen2_8016e37c_gx(uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm) {
|
||||
if (sp >= static_cast<uint32_t>(GX_MAX_TEXGENSRC)) {
|
||||
uint32_t pc = 0;
|
||||
uint32_t lr = 0;
|
||||
@@ -212,24 +245,24 @@ extern "C" void GX__SetTexCoordGen2_8016e37c(uint32_t dc, uint32_t f, uint32_t s
|
||||
}
|
||||
GXSetTexCoordGen2((GXTexCoordID)dc, (GXTexGenType)f, (GXTexGenSrc)sp, m, (GXBool)n, pm);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016e37c, GX__SetTexCoordGen2_8016e37c, (uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm), (dc, f, sp, m, n, pm));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016e37c, GX__SetTexCoordGen2_8016e37c, (uint32_t dc, uint32_t f, uint32_t sp, uint32_t m, uint32_t n, uint32_t pm), (dc, f, sp, m, n, pm));
|
||||
|
||||
extern "C" void GX__EnableTexOffsets_8016f37c(uint32_t coord, uint32_t lineEnable, uint32_t pointEnable) {
|
||||
static void GX__EnableTexOffsets_8016f37c_gx(uint32_t coord, uint32_t lineEnable, uint32_t pointEnable) {
|
||||
GXEnableTexOffsets(static_cast<GXTexCoordID>(coord),
|
||||
lineEnable ? GX_TRUE : GX_FALSE,
|
||||
pointEnable ? GX_TRUE : GX_FALSE);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f37c, GX__EnableTexOffsets_8016f37c, (uint32_t coord, uint32_t lineEnable, uint32_t pointEnable), (coord, lineEnable, pointEnable));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f37c, GX__EnableTexOffsets_8016f37c, (uint32_t coord, uint32_t lineEnable, uint32_t pointEnable), (coord, lineEnable, pointEnable));
|
||||
|
||||
extern "C" void GX__SetLineWidth_8016f314(uint32_t width, uint32_t texOffsets) {
|
||||
static void GX__SetLineWidth_8016f314_gx(uint32_t width, uint32_t texOffsets) {
|
||||
GXSetLineWidth(static_cast<u8>(width), static_cast<GXTexOffset>(texOffsets));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f314, GX__SetLineWidth_8016f314, (uint32_t width, uint32_t texOffsets), (width, texOffsets));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f314, GX__SetLineWidth_8016f314, (uint32_t width, uint32_t texOffsets), (width, texOffsets));
|
||||
|
||||
extern "C" void GX__SetPointSize_8016f348(uint32_t pointSize, uint32_t texOffsets) {
|
||||
static void GX__SetPointSize_8016f348_gx(uint32_t pointSize, uint32_t texOffsets) {
|
||||
GXSetPointSize(static_cast<u8>(pointSize), static_cast<GXTexOffset>(texOffsets));
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f348, GX__SetPointSize_8016f348, (uint32_t pointSize, uint32_t texOffsets), (pointSize, texOffsets));
|
||||
GX_DEFERRED_OVERRIDE_VOID(8016f348, GX__SetPointSize_8016f348, (uint32_t pointSize, uint32_t texOffsets), (pointSize, texOffsets));
|
||||
|
||||
// ============================================================================
|
||||
// Begin/End Drawing
|
||||
@@ -284,6 +317,13 @@ void ServiceDeferredTimingDuringGxWork() {
|
||||
}
|
||||
}
|
||||
|
||||
void GX__Begin_gx(uint32_t t, uint32_t vf, uint32_t nv) {
|
||||
ApplyAuroraVtxStateForBegin(static_cast<GXVtxFmt>(vf));
|
||||
g_hleGxState.currentVtxFmt=(GXVtxFmt)vf; g_hleGxState.currentPrim=(GXPrimitive)t;
|
||||
g_hleGxState.vertsRemaining=nv; g_hleGxState.inBegin=true; g_hleGxState.auroraBeginCalled=false;
|
||||
g_hleGxState.fifoReadOffset = 0;
|
||||
g_hleGxState.fifoByteCount=0; g_hleGxState.ResetVertex();
|
||||
}
|
||||
extern "C" void GX__Begin_8016f0f0(uint32_t t, uint32_t vf, uint32_t nv) {
|
||||
if(IsDisplayListActive()){
|
||||
WriteDisplayListData((u8)(t|vf), 1);
|
||||
@@ -291,21 +331,19 @@ extern "C" void GX__Begin_8016f0f0(uint32_t t, uint32_t vf, uint32_t nv) {
|
||||
return;
|
||||
}
|
||||
ServiceDeferredTimingDuringGxWork();
|
||||
ApplyAuroraVtxStateForBegin(static_cast<GXVtxFmt>(vf));
|
||||
g_hleGxState.currentVtxFmt=(GXVtxFmt)vf; g_hleGxState.currentPrim=(GXPrimitive)t;
|
||||
g_hleGxState.vertsRemaining=nv; g_hleGxState.inBegin=true; g_hleGxState.auroraBeginCalled=false;
|
||||
g_hleGxState.fifoReadOffset = 0;
|
||||
g_hleGxState.fifoByteCount=0; g_hleGxState.ResetVertex();
|
||||
GxThread::Post(&GX__Begin_gx, t, vf, nv);
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(8016f0f0, GX__Begin_8016f0f0, (uint32_t t, uint32_t vf, uint32_t nv), (t, vf, nv));
|
||||
|
||||
extern "C" void GX__End_80044b30() { g_hleGxState.inBegin=false; GXEnd(); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80044b30, GX__End_80044b30, (), ());
|
||||
static void GX__End_80044b30_gx() { g_hleGxState.inBegin=false; GXEnd(); }
|
||||
GX_DEFERRED_OVERRIDE_VOID(80044b30, GX__End_80044b30, (), ());
|
||||
|
||||
extern "C" void GX__End_80048c30() { GX__End_80044b30(); }
|
||||
PPC_NATIVE_OVERRIDE_VOID(80048c30, GX__End_80048c30, (), ());
|
||||
|
||||
extern "C" void GX__DrawSphere_80172a30(uint32_t numMajor, uint32_t numMinor) {
|
||||
// Runs whole on the GX thread: it edits the parser's descriptor state and
|
||||
// restores it, so the game-side mirror is unchanged on return.
|
||||
static void GX__DrawSphere_80172a30_gx(uint32_t numMajor, uint32_t numMinor) {
|
||||
constexpr uint32_t kAttrCount = 26;
|
||||
constexpr float kSphereRadius = 1.0f;
|
||||
constexpr float kPi = 3.14159265358979323846f;
|
||||
@@ -330,13 +368,13 @@ extern "C" void GX__DrawSphere_80172a30(uint32_t numMajor, uint32_t numMinor) {
|
||||
g_hleGxState.InvalidateVtxLayoutHash();
|
||||
GXClearVtxDesc();
|
||||
|
||||
GX__SetVtxDesc_8016d3a4(GX_VA_POS, GX_DIRECT);
|
||||
GX__SetVtxDesc_8016d3a4(GX_VA_NRM, GX_DIRECT);
|
||||
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, GX_VA_POS, GX_POS_XYZ, GX_F32, 0);
|
||||
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, GX_VA_NRM, GX_NRM_XYZ, GX_F32, 0);
|
||||
GX__SetVtxDesc_8016d3a4_gx(GX_VA_POS, GX_DIRECT);
|
||||
GX__SetVtxDesc_8016d3a4_gx(GX_VA_NRM, GX_DIRECT);
|
||||
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, GX_VA_POS, GX_POS_XYZ, GX_F32, 0);
|
||||
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, GX_VA_NRM, GX_NRM_XYZ, GX_F32, 0);
|
||||
if (hadTex0) {
|
||||
GX__SetVtxDesc_8016d3a4(GX_VA_TEX0, GX_DIRECT);
|
||||
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, GX_VA_TEX0, GX_TEX_ST, GX_F32, 0);
|
||||
GX__SetVtxDesc_8016d3a4_gx(GX_VA_TEX0, GX_DIRECT);
|
||||
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, GX_VA_TEX0, GX_TEX_ST, GX_F32, 0);
|
||||
}
|
||||
|
||||
const float majorStep = kPi / static_cast<float>(numMajor);
|
||||
@@ -400,12 +438,12 @@ extern "C" void GX__DrawSphere_80172a30(uint32_t numMajor, uint32_t numMinor) {
|
||||
for (uint32_t attr = 0; attr < kAttrCount; ++attr) {
|
||||
const GXAttrType type = savedVtxDesc[attr];
|
||||
if (type != GX_NONE) {
|
||||
GX__SetVtxDesc_8016d3a4(attr, static_cast<uint32_t>(type));
|
||||
GX__SetVtxDesc_8016d3a4_gx(attr, static_cast<uint32_t>(type));
|
||||
}
|
||||
}
|
||||
for (uint32_t attr = GX_VA_POS; attr <= GX_VA_TEX7; ++attr) {
|
||||
const VtxAttrFmt& fmt = savedFmt3[attr];
|
||||
GX__SetVtxAttrFmt_8016dc68(GX_VTXFMT3, attr, static_cast<uint32_t>(fmt.cnt), static_cast<uint32_t>(fmt.type), fmt.frac);
|
||||
GX__SetVtxAttrFmt_8016dc68_gx(GX_VTXFMT3, attr, static_cast<uint32_t>(fmt.cnt), static_cast<uint32_t>(fmt.type), fmt.frac);
|
||||
}
|
||||
}
|
||||
PPC_NATIVE_OVERRIDE_VOID(80172a30, GX__DrawSphere_80172a30, (uint32_t numMajor, uint32_t numMinor), (numMajor, numMinor));
|
||||
GX_DEFERRED_OVERRIDE_VOID(80172a30, GX__DrawSphere_80172a30, (uint32_t numMajor, uint32_t numMinor), (numMajor, numMinor));
|
||||
+83
-25
@@ -4,6 +4,10 @@
|
||||
#include "guest_interrupt_context.h"
|
||||
#include "ppc_runtime.h"
|
||||
#include "aurora_events.h"
|
||||
#include "gx_thread.h"
|
||||
#include <aurora/imgui.h>
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include "settings_overlay.h"
|
||||
#include "fiber_manager.h"
|
||||
#include "platform/host_platform.h"
|
||||
@@ -48,6 +52,16 @@ extern "C" int32_t OS__RestoreInterrupts_801a65d4(int32_t level);
|
||||
std::atomic_bool g_auroraFrameActive{false};
|
||||
std::atomic_bool g_auroraFrameHadWork{false};
|
||||
|
||||
// GX-thread routine (inline when the thread is off): the aurora frame flags are
|
||||
// written only where aurora's producer runs.
|
||||
static void GxBeginAuroraFrameIfIdle_gx() {
|
||||
if (!g_auroraFrameActive.load(std::memory_order_acquire)) {
|
||||
if (BeginAuroraFrame()) {
|
||||
g_auroraFrameActive.store(true, std::memory_order_release);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
using namespace std::chrono_literals;
|
||||
@@ -328,12 +342,10 @@ void AdvanceRetrace(CpuContext* ctx, Clock::time_point retraceStamp, bool servic
|
||||
// Process window events (but don't present - that happens in GXCopyDisp)
|
||||
UpdateAuroraAndProcessEvents();
|
||||
|
||||
// Start a new Aurora frame if one isn't already active
|
||||
if (!g_auroraFrameActive.load(std::memory_order_acquire)) {
|
||||
if (BeginAuroraFrame()) {
|
||||
g_auroraFrameActive.store(true, std::memory_order_release);
|
||||
}
|
||||
}
|
||||
// Start a new Aurora frame if one isn't already active. The attempt runs
|
||||
// where aurora's producer runs; the viewport policy stays on this thread.
|
||||
GxThread::Post(&GxBeginAuroraFrameIfIdle_gx);
|
||||
ApplyPendingMkwDynamicAspectSurface();
|
||||
// XR runtime loss and encoded-work stalls request producer-owned
|
||||
// teardown. Service it on every retrace, including startup, pause, and
|
||||
// minimized-window paths that may never reach a successful present.
|
||||
@@ -522,24 +534,65 @@ void PaceToRetraceBoundary(Clock::time_point deadline) {
|
||||
// pacing thread: the anchor only makes sense against the recorded camera of
|
||||
// this exact frame, and the pacing thread does not know which frame its packet
|
||||
// will be paired with.
|
||||
void PublishVrSceneAnchor() {
|
||||
struct SceneAnchorPublication {
|
||||
std::array<float, 12> anchor{};
|
||||
bool valid = false;
|
||||
};
|
||||
|
||||
SceneAnchorPublication PublishVrSceneAnchor() {
|
||||
static bool s_engaged = false;
|
||||
// Compute the anchor here rather than at the race draw boundary: the
|
||||
// scene's camera matrix for this frame is only set once the draws run, so
|
||||
// reading it earlier pairs a stale camera with a current kart pose.
|
||||
mkw::vr::MkwVRFirstPersonCommit();
|
||||
const mkw::vr::FirstPersonAnchor anchor = mkw::vr::MkwVRFirstPersonGetAnchor();
|
||||
aurora_set_stereo_scene_anchor(anchor.valid ? anchor.anchor_from_scene.data() : nullptr);
|
||||
SceneAnchorPublication publication;
|
||||
publication.valid = anchor.valid;
|
||||
if (anchor.valid) {
|
||||
std::copy(anchor.anchor_from_scene.begin(), anchor.anchor_from_scene.end(), publication.anchor.begin());
|
||||
}
|
||||
|
||||
const bool engaged = anchor.valid;
|
||||
if (engaged == s_engaged) {
|
||||
return;
|
||||
return publication;
|
||||
}
|
||||
s_engaged = engaged;
|
||||
// The world scale changes with the camera, and the virtual screen's metres
|
||||
// are converted at that scale, so the two have to move together.
|
||||
mkw::vr::MkwVRPolicySetFirstPersonEngaged(engaged);
|
||||
settings_overlay::RefreshVrHudVirtualScreen();
|
||||
return publication;
|
||||
}
|
||||
|
||||
// Everything the seal needs, latched on the game thread and applied by the GX
|
||||
// thread in one record, so the schedule, anchor, tag and overlay all belong to
|
||||
// the frame whose GX commands precede it in the ring.
|
||||
struct GxPresentRecord {
|
||||
uint64_t scheduleBaseNanos = 0;
|
||||
uint64_t scheduleIntervalNanos = 0;
|
||||
std::array<float, 12> anchor{};
|
||||
bool anchorValid = false;
|
||||
bool reportPaced = false;
|
||||
bool paced = false;
|
||||
uint32_t localPlayerCount = 1;
|
||||
uint64_t contentTag = 0;
|
||||
void* imguiFrame = nullptr;
|
||||
};
|
||||
|
||||
void GxPresent_gx(GxPresentRecord record) {
|
||||
if (record.reportPaced) {
|
||||
aurora_report_producer_paced(record.paced);
|
||||
}
|
||||
aurora_set_present_schedule(record.scheduleBaseNanos, record.scheduleIntervalNanos);
|
||||
aurora_set_stereo_scene_anchor(record.anchorValid ? record.anchor.data() : nullptr);
|
||||
aurora_set_stereo_local_player_count(record.localPlayerCount);
|
||||
aurora_end_frame_ex(record.contentTag, record.imguiFrame);
|
||||
g_auroraFrameActive.store(false, std::memory_order_release);
|
||||
g_auroraFrameHadWork.store(false, std::memory_order_release);
|
||||
// Pre-warm the next frame so subsequent GX work has a valid frame context.
|
||||
if (BeginAuroraFrame()) {
|
||||
g_auroraFrameActive.store(true, std::memory_order_release);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -556,6 +609,7 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) {
|
||||
} sequenceGuard;
|
||||
Clock::time_point paceDeadline{};
|
||||
bool paceThisFrame = false;
|
||||
GxPresentRecord record;
|
||||
if (paceToRetrace) {
|
||||
uint64_t baseNanos = 0;
|
||||
uint64_t intervalNanos = 0;
|
||||
@@ -586,7 +640,8 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) {
|
||||
// cadence, and zero only happens when production outruns VI. "Kept up" means <=1 retrace
|
||||
// elapsed; 2+ means a boundary was missed, so Aurora seals that frame without its interpolated
|
||||
// slots (a windowed backstop lowers the slot target only under sustained overload).
|
||||
aurora_report_producer_paced(retracesElapsed <= 1);
|
||||
record.reportPaced = true;
|
||||
record.paced = retracesElapsed <= 1;
|
||||
// Stamp the sealed frame's presentation schedule so Aurora paces interpolated slots against
|
||||
// this same VI timeline. Anchor to the NEXT retrace boundary, not the period just produced,
|
||||
// since slots anchored to the current period would already be expired by seal time. Encoding
|
||||
@@ -599,47 +654,50 @@ void VI_HLE_PresentFrame(bool presentedXfb, bool paceToRetrace) {
|
||||
anchorNanos = s_lastPresentAnchorNanos + intervalNanos;
|
||||
}
|
||||
s_lastPresentAnchorNanos = anchorNanos;
|
||||
aurora_set_present_schedule(anchorNanos, intervalNanos);
|
||||
record.scheduleBaseNanos = anchorNanos;
|
||||
record.scheduleIntervalNanos = intervalNanos;
|
||||
} else {
|
||||
// Retrace-context presents (VI black, boot) have no display period of
|
||||
// their own to subdivide; present as soon as the frame is ready. The
|
||||
// schedule grid is gone, so the anchor cursor must not constrain the
|
||||
// next paced frame.
|
||||
s_lastPresentAnchorNanos = 0;
|
||||
aurora_set_present_schedule(0, 0);
|
||||
}
|
||||
|
||||
mkw::vr::OpenXRServiceProducerFrameBoundary();
|
||||
PublishVrSceneAnchor();
|
||||
const SceneAnchorPublication anchor = PublishVrSceneAnchor();
|
||||
record.anchor = anchor.anchor;
|
||||
record.anchorValid = anchor.valid;
|
||||
// Latch the current policy safety state into this exact Aurora job. The
|
||||
// asynchronous worker may ask for an XR packet after the guest has already
|
||||
// begun the next frame, so immersive replay is accepted only when both
|
||||
// tags match.
|
||||
const auto vrPolicy = mkw::vr::MkwVRPolicyGetSnapshot();
|
||||
aurora_set_stereo_local_player_count(
|
||||
record.localPlayerCount =
|
||||
vrPolicy.presentation == mkw::vr::VRPresentationMode::ImmersiveRace
|
||||
? vrPolicy.scene.local_player_count : 1);
|
||||
aurora_end_frame_tagged(vrPolicy.content_tag);
|
||||
? vrPolicy.scene.local_player_count : 1;
|
||||
record.contentTag = vrPolicy.content_tag;
|
||||
// The overlay drew into the game thread's ImGui frame; close it and hand a
|
||||
// copy of its draw data to the seal.
|
||||
record.imguiFrame = aurora_imgui_host_frame_end();
|
||||
GxThread::Post(&GxPresent_gx, record);
|
||||
if (paceThisFrame) {
|
||||
PaceToRetraceBoundary(paceDeadline);
|
||||
std::lock_guard<std::mutex> lock(g_viMutex);
|
||||
s_lastPacedRetraceCount = g_vi.retraceCount;
|
||||
}
|
||||
settings_overlay::AdvancePresentedFrame();
|
||||
g_auroraFrameActive.store(false, std::memory_order_release);
|
||||
g_auroraFrameHadWork.store(false, std::memory_order_release);
|
||||
if (presentedXfb) {
|
||||
std::lock_guard<std::mutex> lock(g_viMutex);
|
||||
g_vi.hasValidXfb = false;
|
||||
g_vi.readyXfb = 0;
|
||||
}
|
||||
// Pre-warm the next frame so subsequent GX work has a valid frame context.
|
||||
{
|
||||
UpdateAuroraAndProcessEvents();
|
||||
if (BeginAuroraFrame()) {
|
||||
g_auroraFrameActive.store(true, std::memory_order_release);
|
||||
}
|
||||
}
|
||||
// The GX thread begins the next frame right after the seal; the window's
|
||||
// thread pumps events, applies any surface change and starts the next
|
||||
// ImGui frame for the overlay.
|
||||
UpdateAuroraAndProcessEvents();
|
||||
ApplyPendingMkwDynamicAspectSurface();
|
||||
aurora_imgui_host_frame_begin();
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
@@ -54,6 +54,8 @@
|
||||
#include "abi_bridge.h"
|
||||
#include "guest_flat_memory.h"
|
||||
#include "gx_guest_write.h"
|
||||
#include "gx_thread.h"
|
||||
#include <aurora/imgui.h>
|
||||
#include "memory.h"
|
||||
#include "system_bridge.h"
|
||||
#include "ppc_runtime.h"
|
||||
@@ -99,6 +101,14 @@ void ServiceGuestTimingDuringAuroraFrameWait() {
|
||||
Audio_HLE_PollDeferred();
|
||||
}
|
||||
|
||||
void GxThreadFrameLog(char* buffer, uint32_t bufferSize, double windowSeconds, uint32_t frames) {
|
||||
if (!GxThread::Enabled()) {
|
||||
return;
|
||||
}
|
||||
const std::string line = GxThread::FormatStatsAndReset(windowSeconds, frames);
|
||||
std::snprintf(buffer, bufferSize, "%s", line.c_str());
|
||||
}
|
||||
|
||||
#if defined(_WIN32)
|
||||
int __cdecl WindowsCrtReportHook(int reportType, char* message, int* returnValue) {
|
||||
if (returnValue) {
|
||||
@@ -1497,6 +1507,19 @@ int RuntimeMain(int argc, char** argv) {
|
||||
}
|
||||
aurora_set_frame_worker_wait_callback(ServiceGuestTimingDuringAuroraFrameWait);
|
||||
GxGuestWrite::InstallAuroraHooks();
|
||||
GxThread::Configure(RuntimeConfigFile::GxThread());
|
||||
GxThread::SetWaitCallback(ServiceGuestTimingDuringAuroraFrameWait);
|
||||
GxThread::Start();
|
||||
if (GxThread::Enabled()) {
|
||||
// The GX thread is aurora's producer: the game thread never waits
|
||||
// inside aurora any more, and it pumps SDL itself (aurora_update).
|
||||
aurora_set_frame_worker_wait_callback(nullptr);
|
||||
aurora_set_host_event_pump(true);
|
||||
}
|
||||
aurora_set_frame_log_callback(GxThreadFrameLog);
|
||||
// The desktop overlay's ImGui frames belong to this thread from the
|
||||
// first frame on; the seal replays a copy of their draw data.
|
||||
aurora_imgui_host_frame_begin();
|
||||
UpdateMkwDynamicAspectSurface(auroraInfo.windowSize.native_fb_width,
|
||||
auroraInfo.windowSize.native_fb_height);
|
||||
settings_overlay::InitializeRuntimeSettings();
|
||||
@@ -1549,6 +1572,7 @@ int RuntimeMain(int argc, char** argv) {
|
||||
// Shutdown fiber system
|
||||
Fiber::GuestFiberManager::Shutdown();
|
||||
WindowPlacementPersistence::Flush(true);
|
||||
GxThread::Stop();
|
||||
mkw::vr::OpenXRShutdownBeforeAurora();
|
||||
aurora_shutdown();
|
||||
DiscordPresence::Shutdown();
|
||||
@@ -1568,6 +1592,7 @@ int RuntimeMain(int argc, char** argv) {
|
||||
SetRuntimeExitCodeImpl(1);
|
||||
Fiber::GuestFiberManager::Shutdown();
|
||||
WindowPlacementPersistence::Flush(true);
|
||||
GxThread::Stop();
|
||||
mkw::vr::OpenXRShutdownBeforeAurora();
|
||||
aurora_shutdown();
|
||||
DiscordPresence::Shutdown();
|
||||
@@ -1581,6 +1606,7 @@ int RuntimeMain(int argc, char** argv) {
|
||||
SetRuntimeExitCodeImpl(1);
|
||||
Fiber::GuestFiberManager::Shutdown();
|
||||
WindowPlacementPersistence::Flush(true);
|
||||
GxThread::Stop();
|
||||
mkw::vr::OpenXRShutdownBeforeAurora();
|
||||
aurora_shutdown();
|
||||
DiscordPresence::Shutdown();
|
||||
|
||||
@@ -2143,9 +2143,9 @@ void ReleaseControllers() noexcept {
|
||||
}
|
||||
|
||||
void Draw() noexcept {
|
||||
// Wait for the frame worker's DONE phase: it has replayed the previous frame's ImGui draw lists
|
||||
// and started the next ImGui frame, so all overlay callers can now safely issue ImGui commands.
|
||||
aurora_wait_for_frame_worker();
|
||||
// The overlay draws into the game thread's own ImGui frame
|
||||
// (aurora_imgui_host_frame_begin); the worker replays a copy of the draw
|
||||
// data, so there is nothing to wait for here.
|
||||
// Explain fallback without interrupting gameplay or capturing input.
|
||||
static std::string shownXrError;
|
||||
static double xrNoticeUntil = 0.0;
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#include "vr/openxr_integration.h"
|
||||
|
||||
#include "runtime_config.h"
|
||||
#include "gx_thread.h"
|
||||
#include "runtime_log.h"
|
||||
#include "vr/mkw_vr_first_person.h"
|
||||
#include "vr/mkw_vr_policy.h"
|
||||
@@ -473,6 +474,9 @@ public:
|
||||
|
||||
void Shutdown() noexcept {
|
||||
teardown_requested_.store(false, std::memory_order_release);
|
||||
// Called on the game thread: aurora's producer (the GX thread) must be
|
||||
// idle before the frame worker is quiesced.
|
||||
GxThread::Drain();
|
||||
// Stop idle replays before draining; no new worker job may race provider removal.
|
||||
{
|
||||
std::lock_guard lock(interpolation_mutex_);
|
||||
@@ -637,6 +641,16 @@ private:
|
||||
<< (hinted ? "set" : "refused") << std::endl;
|
||||
return true;
|
||||
}
|
||||
bool RegisterGxThread() {
|
||||
const uint32_t thread_id = GxThread::NativeThreadId();
|
||||
if (thread_id == 0 || runtime_ == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const bool hinted = OpenXRAndroidRegisterThreadId(*runtime_, OpenXRAndroidThreadType::RendererWorker, thread_id);
|
||||
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: Android thread hint for the GX thread "
|
||||
<< (hinted ? "set" : "refused") << std::endl;
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Asks the runtime for the configured performance level in both domains. Standalone
|
||||
@@ -689,6 +703,7 @@ private:
|
||||
// frame worker, which submits the GPU work, are the ones that matter; this thread only
|
||||
// paces.
|
||||
bool worker_registered = false;
|
||||
bool gx_registered = false;
|
||||
if (runtime_ != nullptr) {
|
||||
const bool pacing_hinted =
|
||||
OpenXRAndroidRegisterThread(*runtime_, OpenXRAndroidThreadType::RendererWorker);
|
||||
@@ -700,6 +715,7 @@ private:
|
||||
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: Android thread hints: game " << (game_hinted ? "set" : "refused")
|
||||
<< ", pacing " << (pacing_hinted ? "set" : "refused") << std::endl;
|
||||
worker_registered = RegisterAuroraFrameWorkerThread();
|
||||
gx_registered = RegisterGxThread();
|
||||
}
|
||||
#endif
|
||||
ApplyPerformanceLevel();
|
||||
@@ -716,6 +732,9 @@ private:
|
||||
if (!worker_registered) {
|
||||
worker_registered = RegisterAuroraFrameWorkerThread();
|
||||
}
|
||||
if (!gx_registered) {
|
||||
gx_registered = RegisterGxThread();
|
||||
}
|
||||
#endif
|
||||
const OpenXREventStatus events = runtime_->PollEvents();
|
||||
const bool session_active = runtime_->IsSessionRunning();
|
||||
|
||||
Reference in new issue
Block a user