Enhance texture management and profiling in GX thread; add reinitialization flag and optimize consumer behavior

This commit is contained in:
iChris4 committed 2026-09-20 17:10:00 +02:00
1 parent 6a1641e0b7
commit c4c7078afc
5 files changed
+145 -17

No files matched your search

+4
View File
@@ -28,6 +28,7 @@
#include "gfx/pipeline_cache.hpp"
#endif
#if defined(__ANDROID__)
#include <pthread.h>
#include <unistd.h>
#endif
#include "system_info.hpp"
@@ -355,6 +356,9 @@ void frame_worker_main() noexcept {
}
#if defined(__ANDROID__)
g_frameWorkerNativeThreadId.store(static_cast<uint32_t>(gettid()), std::memory_order_release);
// A thread inherits its creator's name, and the producer that starts this
// worker may itself be a named thread; profiles should tell the two apart.
pthread_setname_np(pthread_self(), "aurora worker");
#endif
#ifdef AURORA_ENABLE_GX
+17
View File
@@ -723,6 +723,23 @@ waited under 0.1 ms per frame in its two `GXDrawDone` drains and never for
ring space, and the GX thread was 25 to 40% busy. Retro Rewind's menus were
unaffected (prewarm 5.2 s, 60 fps).
Two things the first day on it taught. The Retro Rewind menu with the blurred
background fell to 14 to 18 fps, with the GX thread on or off, and the
per-record profile that the `fpslog` line now carries (`costliest:`) put it
all in the FIFO records: the game re-initialises its capture texture objects
every frame, and the split had kept one aurora object per guest object alive
across those re-initialisations, so `GXInitTexObjData` kept incrementing
`texDataVersion`, which is part of aurora's static upload key, and every
frame converted every such texture again (`convert_texture` 18% of the
thread). A guest `GXInitTexObj` now rebuilds the aurora object, as it always
had, so the version restarts and the upload cache hits. Second, that menu
calls `GXDrawDone` 22 to 24 times per frame (the base main menu 9 times),
and each drain cost about 1.5 ms while the game thread slept on a condition
variable: both the drain and the idle consumer now spin for a few hundred
microseconds before blocking, with a sequentially consistent sleep handshake,
and the 22 drains cost 2.6 ms per frame in total; that screen runs at 60 with
the thread on.
Verified on device since: the menus on the virtual screen, controller input
(the user has driven races), and an immersive Grand Prix start with all 12
racers rendering correctly. Not yet verified: stereo comfort and scale,
+5
View File
@@ -101,6 +101,10 @@ extern float g_projectionVector[7];
struct TexObjMeta : GxTextureBindingContract::SamplerState {
uint32_t userData = 0;
bool needsUpload = true;
// GXInitTexObj/GXInitTexObjCI ran since the last load: the aurora object is
// rebuilt from scratch on the next load, as it always was, so its data
// version restarts and aurora's upload cache keys repeat across frames.
bool reinitPending = false;
};
// --- Host-side texture object storage (audit F9) ---
@@ -384,6 +388,7 @@ struct GxTexObjLoad {
uint32_t objAddr = 0;
uint32_t tid = 0;
bool upload = false;
bool reinit = false;
};
void GxHostLoadTexObj_gx(GxTexObjLoad load);
void GxHostBindPlaceholder_gx(uint32_t tid);
+8 -3
View File
@@ -251,7 +251,7 @@ extern "C" void GX__InitTexObj_801707f8(uint32_t oa, uint32_t da, uint32_t w, ui
}
const uint32_t canonicalDataAddr = CanonicalizeGxMainRamAddress(da);
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.userData=0; meta.needsUpload=true;
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.userData=0; meta.needsUpload=true; meta.reinitPending=true;
// Also write to guest memory so reads work
WriteGuestTexObj(oa, canonicalDataAddr, (u16)w, (u16)h, f, ws, wt, m != 0, false, 0);
}
@@ -291,7 +291,7 @@ extern "C" void GX__InitTexObjCI_80170a04(uint32_t oa, uint32_t da, uint32_t w,
}
const uint32_t canonicalDataAddr = CanonicalizeGxMainRamAddress(da);
std::lock_guard<std::mutex> guard(g_texObjMutex); TexObjMeta& meta = GetTexObjMeta(oa);
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.tlut=tl; meta.userData=0; meta.needsUpload=true;
meta.dataAddr=canonicalDataAddr; meta.width=(u16)w; meta.height=(u16)h; meta.format=f; meta.wrapS=ws; meta.wrapT=wt; meta.mipmap=(m!=0); meta.tlut=tl; meta.userData=0; meta.needsUpload=true; meta.reinitPending=true;
// Also write to guest memory so reads work
WriteGuestTexObj(oa, canonicalDataAddr, (u16)w, (u16)h, f, ws, wt, m != 0, true, tl);
// GXInitTexObjCI clears bit1 in the flags byte; keep guest memory consistent.
@@ -528,7 +528,9 @@ extern "C" void GX__LoadTexObj_80170f2c(uint32_t oa, uint32_t tid) {
load.objAddr = oa;
load.tid = tid;
load.upload = meta.needsUpload;
load.reinit = meta.reinitPending;
meta.needsUpload = false;
meta.reinitPending = false;
load.meta = meta;
// Write through GetTexObjMeta so the DCStoreRange interval index is
// told this entry's backing may have moved.
@@ -627,7 +629,10 @@ void GxHostLoadTexObj_gx(GxTexObjLoad load) {
const uint32_t size = GXGetTexBufferSize(meta.width, meta.height, meta.format, (GXBool)meta.mipmap, maxLod);
GxHostTexObjEntry& entry = g_gxHostTexObjs[oa];
GXTexObj* obj = entry.host.constructed ? entry.host.PublicPtr() : nullptr;
const bool needsInit = obj == nullptr || !SameTexObjBuildMeta(entry.cached, meta);
// A guest GXInitTexObj always rebuilt the aurora object before this split;
// keeping that resets texDataVersion, so aurora's static upload cache keeps
// hitting for textures the game re-initialises every frame (menu captures).
const bool needsInit = obj == nullptr || load.reinit || !SameTexObjBuildMeta(entry.cached, meta);
bool textureDataUploaded = false;
try {
if (needsInit) {
+111 -14
View File
@@ -8,9 +8,11 @@
#include <exception>
#include <mutex>
#include <thread>
#include <algorithm>
#if defined(_WIN32)
#include <windows.h>
#else
#include <dlfcn.h>
#include <pthread.h>
#include <unistd.h>
#if defined(__linux__)
@@ -30,7 +32,11 @@ constexpr uint32_t kRingBytes = 16u << 20;
constexpr uint32_t kRingMask = kRingBytes - 1u;
constexpr uint32_t kHeaderBytes = 16;
constexpr uint32_t kFifoChunkBytes = 8192;
constexpr uint32_t kConsumerSpinIterations = 4000;
// An empty ring is usually a gap of microseconds between two bursts of the
// same frame, so the consumer spins that long before it pays a futex sleep;
// a drain likewise spins before blocking, since its backlog is normally short.
constexpr std::chrono::microseconds kConsumerSpinBudget{50};
constexpr std::chrono::microseconds kDrainSpinBudget{300};
struct Header {
uint64_t invoke;
@@ -79,6 +85,66 @@ uint64_t g_statQueuePeakBytes = 0;
std::atomic<uint64_t> g_statBusyNs{0};
std::atomic<uint64_t> g_statFaults{0};
// Per-record-kind profile for the frame-rate log: keyed by the back function
// (or 1 for FIFO chunks, 2 for fences). Consumer-written; the producer reads
// it when it formats the log line, which is diagnostics, not synchronisation.
struct RecordProfile {
uintptr_t key = 0;
uint64_t ns = 0;
uint64_t count = 0;
};
constexpr size_t kProfileSlots = 128;
constexpr uintptr_t kProfileKeyFifo = 1;
constexpr uintptr_t kProfileKeyFence = 2;
RecordProfile g_profile[kProfileSlots];
void ProfileRecord(uintptr_t key, uint64_t ns) {
size_t slot = static_cast<size_t>(key >> 4) % kProfileSlots;
for (size_t probe = 0; probe < kProfileSlots; ++probe) {
RecordProfile& entry = g_profile[slot];
if (entry.key == key || entry.key == 0) {
entry.key = key;
entry.ns += ns;
++entry.count;
return;
}
slot = (slot + 1) % kProfileSlots;
}
}
// Names a back function for the log: its symbol when the loader knows it,
// otherwise its offset in the module, for llvm-symbolizer on the build's .so.
std::string DescribeProfileKey(uintptr_t key) {
if (key == kProfileKeyFifo) {
return "fifo";
}
if (key == kProfileKeyFence) {
return "fence";
}
char buffer[96];
#if defined(_WIN32)
HMODULE module = nullptr;
if (::GetModuleHandleExW(GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS | GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
reinterpret_cast<LPCWSTR>(key), &module) && module != nullptr) {
std::snprintf(buffer, sizeof(buffer), "+0x%llx",
static_cast<unsigned long long>(key - reinterpret_cast<uintptr_t>(module)));
return buffer;
}
#else
Dl_info info{};
if (dladdr(reinterpret_cast<const void*>(key), &info) != 0) {
if (info.dli_sname != nullptr && info.dli_saddr == reinterpret_cast<const void*>(key)) {
return info.dli_sname;
}
std::snprintf(buffer, sizeof(buffer), "+0x%llx",
static_cast<unsigned long long>(key - reinterpret_cast<uintptr_t>(info.dli_fbase)));
return buffer;
}
#endif
std::snprintf(buffer, sizeof(buffer), "0x%llx", static_cast<unsigned long long>(key));
return buffer;
}
uint64_t ElapsedNs(Clock::time_point since) {
return static_cast<uint64_t>(std::chrono::duration_cast<std::chrono::nanoseconds>(Clock::now() - since).count());
}
@@ -128,7 +194,7 @@ void WaitWithCallback(std::condition_variable& cv, std::atomic<bool>& waitingFla
}
void NotifyConsumer() {
if (g_consumerSleeping.load(std::memory_order_acquire)) {
if (g_consumerSleeping.load(std::memory_order_seq_cst)) {
std::lock_guard<std::mutex> lock(g_mutex);
g_cvData.notify_one();
}
@@ -157,7 +223,9 @@ void PostRecordRaw(detail::Invoke invoke, const void* payload, uint32_t payloadB
std::memcpy(g_ring + offset + kHeaderBytes, payload, payloadBytes);
}
g_localTail += stride;
g_tail.store(g_localTail, std::memory_order_release);
// seq_cst pairs with the consumer's sleeping flag: one of the two sides
// always sees the other's store, so a post never leaves the consumer asleep.
g_tail.store(g_localTail, std::memory_order_seq_cst);
++g_statRecords;
g_statBytes += stride;
const uint64_t queued = g_localTail - g_head.load(std::memory_order_relaxed);
@@ -184,25 +252,30 @@ void ConsumerLoop() {
SetThreadName();
uint64_t head = g_head.load(std::memory_order_relaxed);
uint32_t spins = 0;
Clock::time_point spinStarted{};
while (true) {
const uint64_t tail = g_tail.load(std::memory_order_acquire);
if (head == tail) {
if (g_stop.load(std::memory_order_acquire)) {
break;
}
if (spins < kConsumerSpinIterations) {
if (spins == 0) {
spinStarted = Clock::now();
}
++spins;
if ((spins & 31u) != 0 || Clock::now() - spinStarted < kConsumerSpinBudget) {
std::this_thread::yield();
continue;
}
g_consumerSleeping.store(true, std::memory_order_release);
g_consumerSleeping.store(true, std::memory_order_seq_cst);
{
std::unique_lock<std::mutex> lock(g_mutex);
g_cvData.wait_for(lock, std::chrono::milliseconds(1), [&] {
return g_tail.load(std::memory_order_acquire) != head || g_stop.load(std::memory_order_acquire);
return g_tail.load(std::memory_order_seq_cst) != head || g_stop.load(std::memory_order_acquire);
});
}
g_consumerSleeping.store(false, std::memory_order_release);
g_consumerSleeping.store(false, std::memory_order_seq_cst);
spins = 0;
continue;
}
spins = 0;
@@ -211,6 +284,13 @@ void ConsumerLoop() {
if (header.invoke != 0) {
const auto started = Clock::now();
const uint8_t* payload = g_ring + ((head + kHeaderBytes) & kRingMask);
uintptr_t profileKey = kProfileKeyFifo;
if (header.invoke == reinterpret_cast<uint64_t>(&FenceInvoke)) {
profileKey = kProfileKeyFence;
} else if (header.invoke != reinterpret_cast<uint64_t>(&FifoInvoke)) {
// Every call record starts with the back function's pointer.
std::memcpy(&profileKey, payload, sizeof(profileKey));
}
try {
reinterpret_cast<detail::Invoke>(header.invoke)(payload, header.payloadBytes);
} catch (const std::exception& ex) {
@@ -226,7 +306,9 @@ void ConsumerLoop() {
static_cast<unsigned long long>(faults));
}
}
g_statBusyNs.fetch_add(ElapsedNs(started), std::memory_order_relaxed);
const uint64_t elapsed = ElapsedNs(started);
g_statBusyNs.fetch_add(elapsed, std::memory_order_relaxed);
ProfileRecord(profileKey, elapsed);
}
head += header.stride;
g_head.store(head, std::memory_order_release);
@@ -296,12 +378,15 @@ void Drain() {
const uint64_t sequence = ++g_fenceRequested;
PostRecordRaw(&FenceInvoke, &sequence, sizeof(sequence));
++g_statDrains;
if (g_fenceCompleted.load(std::memory_order_acquire) >= sequence) {
return;
}
const auto started = Clock::now();
while (g_fenceCompleted.load(std::memory_order_acquire) < sequence) {
if (Clock::now() - started >= kDrainSpinBudget) {
WaitWithCallback(g_cvFence, g_producerWaiting,
[&] { return g_fenceCompleted.load(std::memory_order_acquire) >= sequence; });
break;
}
std::this_thread::yield();
}
g_statDrainWaitNs += ElapsedNs(started);
}
@@ -340,8 +425,8 @@ std::string FormatStatsAndReset(double windowSeconds, uint32_t frames) {
const double perFrame = frames != 0 ? 1.0 / static_cast<double>(frames) : 0.0;
const double busyNs = static_cast<double>(g_statBusyNs.exchange(0, std::memory_order_relaxed));
const double busyPercent = windowSeconds > 0.0 ? busyNs / (windowSeconds * 1e9) * 100.0 : 0.0;
char buffer[512];
std::snprintf(buffer, sizeof(buffer),
char buffer[1024];
const int written = std::snprintf(buffer, sizeof(buffer),
"GX thread: %.0f records/frame (%.1f KiB, %.0f FIFO chunks with %.1f KiB), queue peak %.1f KiB; "
"game thread waited %.2f ms/frame for ring space and %.2f ms/frame in %.1f drains/frame; "
"GX thread busy %.1f%%; faults %llu",
@@ -356,7 +441,19 @@ std::string FormatStatsAndReset(double windowSeconds, uint32_t frames) {
static_cast<unsigned long long>(g_statFaults.load(std::memory_order_relaxed)));
g_statRecords = g_statFifoRecords = g_statFifoBytes = g_statBytes = 0;
g_statDrains = g_statDrainWaitNs = g_statSpaceWaitNs = g_statQueuePeakBytes = 0;
return buffer;
// The costliest record kinds of the window, as ms per frame and calls per frame.
RecordProfile top[kProfileSlots];
std::memcpy(top, g_profile, sizeof(top));
std::memset(g_profile, 0, sizeof(g_profile));
std::sort(std::begin(top), std::end(top), [](const RecordProfile& a, const RecordProfile& b) { return a.ns > b.ns; });
std::string line = written > 0 ? std::string(buffer, static_cast<size_t>(std::min<int>(written, sizeof(buffer) - 1))) : std::string();
line += "; costliest:";
for (size_t i = 0; i < 6 && top[i].key != 0; ++i) {
std::snprintf(buffer, sizeof(buffer), " %s %.2f ms x%.0f", DescribeProfileKey(top[i].key).c_str(),
static_cast<double>(top[i].ns) * perFrame / 1e6, static_cast<double>(top[i].count) * perFrame);
line += buffer;
}
return line;
}
namespace detail {