Added performance level configuration for OpenXR runtime and enhance Aurora's frame worker

This commit is contained in:
iChris4 committed 2026-09-19 04:43:47 +02:00
1 parent d296b3dcb3
commit 60c443654f
12 files changed
+292 -54

No files matched your search

+6 -4
View File
@@ -39,10 +39,12 @@ inline uint32_t CanonicalizeGxMainRamAddress(uint32_t addr) noexcept {
namespace GxGuestWrite {
// Granularity matches the GX resource caches' occupancy maps: 64 KiB, so a
// cacheable display list or texture spans only a handful of counters. A false
// bump only costs one re-digest, which is exactly the untracked behaviour.
inline constexpr uint32_t kGranuleShift = 16; // 64 KiB per granule
// 4 KiB per granule, so a cacheable display list or texture spans a few counters and a false
// bump only costs one re-digest, which is exactly the untracked behaviour. It used to be 64 KiB:
// then per-frame guest writes landing beside a static list or texture (a race start streams
// data next to them) bumped the same granule, and every affected list was re-digested and every
// affected texture re-hashed each frame. The table is 320 KiB.
inline constexpr uint32_t kGranuleShift = 12;
inline constexpr uint64_t kTrackedSpan = static_cast<uint64_t>(Memory::kMem2PhysicalEnd);
inline constexpr size_t kGranuleCount = static_cast<size_t>(kTrackedSpan >> kGranuleShift);
+33
View File
@@ -68,6 +68,7 @@ struct RuntimeUserConfig {
std::optional<bool> vrFirstPersonHideDriver;
std::optional<int32_t> vrFirstPersonHiddenModel;
std::optional<std::string> vrFirstPersonRotation;
std::optional<std::string> vrPerformanceLevel;
std::optional<std::string> vrRecenterKey;
std::optional<float> vrLeanBackDegrees;
// F10 > Diagnostics: OpenXR pacing and presentation logging in console.log.
@@ -179,6 +180,16 @@ inline constexpr const char* kVrFirstPersonRotationDefault = "yaw";
inline bool IsSupportedVrFirstPersonRotation(std::string_view value) {
return value == "yaw" || value == "yaw_pitch" || value == "full";
}
// The performance level asked of the OpenXR runtime (XR_EXT_performance_settings) for its CPU and
// GPU domains. Standalone headsets clock their cores by this request: a Quest 3 ran the game
// thread at 1.92 GHz with the runtime's own choice while its fast cores reach 2.36 GHz. "default"
// leaves the runtime's choice; desktop runtimes without the extension ignore the setting.
inline constexpr const char* kVrPerformanceLevelDefault = "boost";
inline bool IsSupportedVrPerformanceLevel(std::string_view value) {
return value == "default" || value == "power_savings" || value == "sustained_low" ||
value == "sustained_high" || value == "boost";
}
// What the desktop window shows while the headset is running: "normal" leaves
// the ordinary desktop view alone, "both", "left" and "right" mirror the
// headset's eyes, and "none" blacks the window out. Matches
@@ -455,6 +466,11 @@ inline void EnsureConfigFile() {
"# horizon, \"yaw_pitch\" adds the kart's climb but no roll, and\n"
"# \"full\" takes the kart's whole orientation so the view banks.\n"
"first_person_rotation = \"yaw\"\n\n"
"# Performance level asked of the headset's runtime for its CPU and\n"
"# GPU: \"boost\", \"sustained_high\", \"sustained_low\", \"power_savings\",\n"
"# or \"default\" to leave the runtime's own choice. Standalone headsets\n"
"# clock their cores by this request; desktop runtimes ignore it.\n"
"performance_level = \"boost\"\n\n"
"# Keyboard shortcut that recenters the VR view, naming the key the\n"
"# way SDL does (F9, Home, Keypad 5, ...). It moves the race view to\n"
"# where you are sitting now and brings the menu screen back upright in\n"
@@ -675,6 +691,10 @@ inline RuntimeUserConfig ParseConfigDocument(const toml::value& document) {
value && IsSupportedVrFirstPersonRotation(*value)) {
config.vrFirstPersonRotation = *value;
}
if (auto value = FindConfigValue<std::string>(document, "vr", "performance_level");
value && IsSupportedVrPerformanceLevel(*value)) {
config.vrPerformanceLevel = *value;
}
if (auto value = FindConfigValue<std::string>(document, "vr", "mirror_view");
value && IsSupportedVrMirrorView(*value)) {
config.vrMirrorView = *value;
@@ -1013,6 +1033,14 @@ inline bool SetVrFirstPersonRotation(std::string value) {
return WriteSetting("vr", "first_person_rotation", FormatString(value));
}
inline bool SetVrPerformanceLevel(std::string value) {
if (!IsSupportedVrPerformanceLevel(value)) {
return false;
}
Mutable().vrPerformanceLevel = value;
return WriteSetting("vr", "performance_level", FormatString(value));
}
inline bool SetVrFirstPersonHiddenModel(int32_t value) {
value = std::clamp(value, -1, 31);
Mutable().vrFirstPersonHiddenModel = value;
@@ -1371,6 +1399,11 @@ inline std::string VrFirstPersonRotation(std::string fallback = kVrFirstPersonRo
return value && IsSupportedVrFirstPersonRotation(*value) ? *value : std::move(fallback);
}
inline std::string VrPerformanceLevel(std::string fallback = kVrPerformanceLevelDefault) {
const auto& value = Get().vrPerformanceLevel;
return value && IsSupportedVrPerformanceLevel(*value) ? *value : std::move(fallback);
}
inline int32_t VrFirstPersonHiddenModel(int32_t fallback = kVrFirstPersonHiddenModelDefault) {
return std::clamp(Get().vrFirstPersonHiddenModel.value_or(fallback), -1, 31);
}
+16 -7
View File
@@ -2,6 +2,7 @@
#include "runtime_log.h"
#include <cstddef>
#include <cstring>
#include <limits>
// --- Texture and TLUT Objects ---
@@ -447,15 +448,23 @@ TexObjMeta MergeGuestTexObjMeta(const TexObjMeta& cached, const TexObjMeta& gues
// Reads the raw 32 guest bytes of a GXTexObj struct. Returns false (and the
// caller must treat the shadow as absent) when the address is unreadable.
// The shadow is only ever compared for equality (ShadowEquals), so it keeps
// the bytes in guest order: one page-table probe and one copy per lookup
// instead of four checked, byte-swapped 64-bit reads. Every draw's texture
// binds go through this on the game thread, so the per-call cost matters.
bool ReadTexObjShadow(uint32_t addr, uint64_t (&out)[4]) noexcept {
try {
out[0] = Memory::Read64(addr + 0x00);
out[1] = Memory::Read64(addr + 0x08);
out[2] = Memory::Read64(addr + 0x10);
out[3] = Memory::Read64(addr + 0x18);
} catch (...) {
return false;
const uint8_t* bytes = MemoryInline::GetPointerFast(addr, sizeof(out));
if (bytes == nullptr) {
try {
bytes = Memory::GetPointer(addr, sizeof(out));
} catch (...) {
return false;
}
if (bytes == nullptr) {
return false;
}
}
std::memcpy(out, bytes, sizeof(out));
return true;
}
+10
View File
@@ -244,6 +244,16 @@ void ServiceDeferredTimingDuringGxWork() {
return;
}
// GX__Begin runs thousands of times a frame, and reading the clock on each one was 2% of the
// game thread on the Quest. The poll has a 1 ms cadence, so sampling the clock on every
// sixteenth call keeps it within a few microseconds of that. A plain static: GX runs on the
// one guest-facing thread, and a thread_local here would cost a resolver call per access.
static uint32_t s_pollCountdown = 0;
if (s_pollCountdown != 0) {
--s_pollCountdown;
return;
}
s_pollCountdown = 15;
const auto now = std::chrono::steady_clock::now();
if (now < g_nextDeferredTimingPoll) {
return;
+34 -2
View File
@@ -369,7 +369,7 @@ public:
#if defined(_WIN32)
config.required_extensions = {"XR_KHR_D3D12_enable"};
config.optional_extensions = {"XR_KHR_win32_convert_performance_counter_time",
"XR_FB_display_refresh_rate"};
"XR_FB_display_refresh_rate", "XR_EXT_performance_settings"};
#else
// Either Vulkan binding extension is acceptable; the backend picks
// whichever the runtime enabled, preferring enable2.
@@ -377,7 +377,7 @@ public:
config.optional_extensions = {"XR_KHR_vulkan_enable2", "XR_KHR_vulkan_enable",
"XR_KHR_convert_timespec_time",
"XR_KHR_android_thread_settings",
"XR_FB_display_refresh_rate"};
"XR_FB_display_refresh_rate", "XR_EXT_performance_settings"};
config.instance_create_next = OpenXRAndroidInstanceCreateNext();
#endif
if (!runtime_->Initialize(config)) {
@@ -401,6 +401,9 @@ public:
if (has_extension("XR_FB_display_refresh_rate")) {
runtime_->LoadFunction("xrGetDisplayRefreshRateFB", &get_display_refresh_rate_);
}
if (has_extension("XR_EXT_performance_settings")) {
runtime_->LoadFunction("xrPerfSettingsSetPerformanceLevelEXT", &set_performance_level_);
}
interpolation_available_.store(convert_display_time_ != nullptr, std::memory_order_release);
if (!backend_->QueryGraphicsRequirements(*runtime_)) {
SetError(backend_->LastError());
@@ -509,6 +512,7 @@ public:
prepared_ = false;
convert_display_time_ = nullptr;
get_display_refresh_rate_ = nullptr;
set_performance_level_ = nullptr;
headset_hz_.store(0, std::memory_order_relaxed);
rendered_fps_.store(0, std::memory_order_relaxed);
interpolation_available_.store(false, std::memory_order_release);
@@ -635,6 +639,32 @@ private:
}
#endif
// Asks the runtime for the configured performance level in both domains. Standalone
// headsets clock their cores by this: a Quest 3 held the game thread at CPU level 4
// (2.2 GHz of a possible 2.36) and the GPU at level 3 with the runtime's own choice. A
// refusal is logged and changes nothing; desktop runtimes rarely offer the extension.
void ApplyPerformanceLevel() {
if (runtime_ == nullptr || set_performance_level_ == nullptr || !runtime_->HasSession()) {
return;
}
const std::string requested = RuntimeConfigFile::VrPerformanceLevel();
XrPerfSettingsLevelEXT level = XR_PERF_SETTINGS_LEVEL_SUSTAINED_HIGH_EXT;
if (requested == "default") {
return;
} else if (requested == "power_savings") {
level = XR_PERF_SETTINGS_LEVEL_POWER_SAVINGS_EXT;
} else if (requested == "sustained_low") {
level = XR_PERF_SETTINGS_LEVEL_SUSTAINED_LOW_EXT;
} else if (requested == "boost") {
level = XR_PERF_SETTINGS_LEVEL_BOOST_EXT;
}
const XrResult cpu = set_performance_level_(runtime_->Session(), XR_PERF_SETTINGS_DOMAIN_CPU_EXT, level);
const XrResult gpu = set_performance_level_(runtime_->Session(), XR_PERF_SETTINGS_DOMAIN_GPU_EXT, level);
RT_LOG(RT_TAG_RUNTIME) << "OpenXR: performance level \"" << requested << "\" CPU "
<< (XR_SUCCEEDED(cpu) ? "set" : "refused") << " (" << cpu << "), GPU "
<< (XR_SUCCEEDED(gpu) ? "set" : "refused") << " (" << gpu << ")" << std::endl;
}
static bool ProvideStereoFrame(uint32_t, AuroraStereoFrame* output, void* userdata) {
auto* self = static_cast<OpenXRIntegration*>(userdata);
if (self == nullptr || output == nullptr) {
@@ -672,6 +702,7 @@ private:
worker_registered = RegisterAuroraFrameWorkerThread();
}
#endif
ApplyPerformanceLevel();
bool fatal = false;
uint32_t consecutive_skips = 0;
bool store_gate_set = false;
@@ -1295,6 +1326,7 @@ private:
std::chrono::steady_clock::time_point timing_start_ = std::chrono::steady_clock::now();
uint32_t timing_submissions_ = 0;
PFN_xrGetDisplayRefreshRateFB get_display_refresh_rate_ = nullptr;
PFN_xrPerfSettingsSetPerformanceLevelEXT set_performance_level_ = nullptr;
#if defined(_WIN32)
using ConvertDisplayTime = XrResult (XRAPI_PTR*)(XrInstance, XrTime, LARGE_INTEGER*);
#else
+32
View File
@@ -566,6 +566,38 @@ OpenXREventStatus OpenXRRuntime::PollEvents() {
Log(OpenXRLogLevel::Warning, message.str());
break;
}
case XR_TYPE_EVENT_DATA_PERF_SETTINGS_EXT: {
// XR_EXT_performance_settings: the runtime reports when a domain's compositing,
// rendering or thermal state moves between normal, warning and impaired. Logged so a
// throttled headset explains a frame-rate drop in the session log.
const auto& perf_event =
*reinterpret_cast<const XrEventDataPerfSettingsEXT*>(&event);
const auto sub_domain = [](XrPerfSettingsSubDomainEXT value) {
switch (value) {
case XR_PERF_SETTINGS_SUB_DOMAIN_COMPOSITING_EXT: return "compositing";
case XR_PERF_SETTINGS_SUB_DOMAIN_RENDERING_EXT: return "rendering";
case XR_PERF_SETTINGS_SUB_DOMAIN_THERMAL_EXT: return "thermal";
default: return "unknown";
}
};
const auto notification = [](XrPerfSettingsNotificationLevelEXT value) {
switch (value) {
case XR_PERF_SETTINGS_NOTIF_LEVEL_NORMAL_EXT: return "normal";
case XR_PERF_SETTINGS_NOTIF_LEVEL_WARNING_EXT: return "warning";
case XR_PERF_SETTINGS_NOTIF_LEVEL_IMPAIRED_EXT: return "impaired";
default: return "unknown";
}
};
std::ostringstream message;
message << "OpenXR performance notification: "
<< (perf_event.domain == XR_PERF_SETTINGS_DOMAIN_CPU_EXT ? "CPU" : "GPU") << " "
<< sub_domain(perf_event.subDomain) << " " << notification(perf_event.fromLevel)
<< " -> " << notification(perf_event.toLevel);
Log(perf_event.toLevel == XR_PERF_SETTINGS_NOTIF_LEVEL_NORMAL_EXT ? OpenXRLogLevel::Info
: OpenXRLogLevel::Warning,
message.str());
break;
}
default:
break;
}