Quest: optimize compilation rules, adjust stack protection, and increase indirect dispatch cache size

This commit is contained in:
iChris4 committed 2026-09-19 03:18:07 +02:00
1 parent 7f2113dbc5
commit d296b3dcb3
4 files changed
+26 -3

No files matched your search

+9 -1
View File
@@ -58,8 +58,11 @@ function(mkw_apply_common_compile_options target)
endfunction()
function(mkw_apply_translated_compile_options target)
# The NDK toolchain adds -fstack-protector-strong to every object. Translated guest code keeps
# its state in guest memory and the CpuContext, so the canaries only cost cycles there.
target_compile_options(${target} PRIVATE
-O2 ${MKW_TRANSLATED_PPC_FP_OPTIONS} -fno-slp-vectorize -w -pipe)
-O2 ${MKW_TRANSLATED_PPC_FP_OPTIONS} -fno-slp-vectorize -w -pipe
$<$<PLATFORM_ID:Android>:-fno-stack-protector>)
endfunction()
function(mkw_configure_object_target target)
@@ -319,6 +322,11 @@ function(mkw_configure_product target)
target_link_libraries(${target} PRIVATE mkw::libco ${CMAKE_DL_LIBS})
elseif(MKW_PLATFORM_ANDROID)
target_link_libraries(${target} PRIVATE mkw::libco android log vulkan ${CMAKE_DL_LIBS})
# A shared library's own calls and data references go through the PLT and GOT unless the
# symbols are bound at link time; with 29,000 translated functions calling each other and
# the runtime, those stubs were 4% of the game thread on the Quest. Nothing interposes
# symbols of the game library, and the JNI and SDL_main exports stay exported.
target_link_options(${target} PRIVATE "-Wl,-Bsymbolic")
endif()
if(MKW_PLATFORM_WINDOWS)
foreach(runtime_dll libc++.dll libunwind.dll)
+4 -1
View File
@@ -303,7 +303,10 @@ struct KnownTypedNativeCpuCall {
// a miss just re-runs the sorted lookup. 512 entries (12 KiB) covers the per-frame indirect
// working set while staying L1/L2 resident, unlike 4096 which would thrash L2. Must stay a
// power of two, the index masks with (size - 1).
inline constexpr size_t kIndirectDispatchCacheEntries = 512;
// Direct-mapped by a hash of the target address. A race keeps a few thousand distinct indirect
// targets live, so 512 entries missed about one call in three on the Quest and fell into the
// table's binary search; 8192 entries are 128 KB per memo and fit the big cores' L2.
inline constexpr size_t kIndirectDispatchCacheEntries = 8192;
// Namespace-scope `inline thread_local`, not function-local `static thread_local`, to avoid a
// thread-static init epoch check on every bctrl (same as g_currentCpuContext in ppc_runtime.h).