Quest: optimize compilation rules, adjust stack protection, and increase indirect dispatch cache size

This commit is contained in:
iChris4 committed 2026-09-19 03:18:07 +02:00
1 parent 7f2113dbc5
commit d296b3dcb3
4 files changed
+26 -3

No files matched your search

+3 -1
View File
@@ -505,7 +505,9 @@ function Invoke-QuestGameBuild {
$escape = { param([string]$p) $p.Replace('$', '$$').Replace(':', '$:').Replace(' ', '$ ') }
[void]$ninjaText.AppendLine('ninja_required_version = 1.10')
[void]$ninjaText.AppendLine("pool translated`n depth = $TranslatedJobs")
[void]$ninjaText.AppendLine("rule cxx`n command = $(ConvertTo-QuotedArgument $ClangCxx) @`$flags -c `$in -o `$out`n description = Compiling `$in")
# Depfiles, so a changed runtime header recompiles the shards that include it instead of
# leaving objects built against the previous kit.
[void]$ninjaText.AppendLine("rule cxx`n command = $(ConvertTo-QuotedArgument $ClangCxx) @`$flags -MD -MF `$out.d -c `$in -o `$out`n depfile = `$out.d`n deps = gcc`n description = Compiling `$in")
[void]$ninjaText.AppendLine("rule asm`n command = $(ConvertTo-QuotedArgument $ClangC) @`$flags -c `$in -o `$out`n description = Assembling `$in")
[void]$ninjaText.AppendLine("rule link`n command = $(ConvertTo-QuotedArgument $ClangCxx) @`$flags -o `$out @`$out.rsp`n rspfile = `$out.rsp`n rspfile_content = `$inputs`n description = Linking `$out")
$objectsOf = @{}
+10
View File
@@ -419,6 +419,16 @@ Android facts this design rests on, all measured on a Quest 3:
SDL3 built shared (or `-DAURORA_SDL3_PROVIDER=system` for the AAR prefab),
tests off, products built as `libmain.so` / `libmain_retro_rewind.so`,
`-mcpu=cortex-a77` (Quest 2's XR2 Gen 1; Quest 3/Pro are supersets).
The products link with `-Wl,-Bsymbolic` and the translated shards compile
with `-fno-stack-protector`: without them every one of the 29,000 translated
functions called its neighbours and the runtime through a PLT stub (26,973
of them, 702 after), which was 4% of the game thread on a Quest 3, and the
NDK's default canaries cost cycles in code whose state lives in guest memory.
With those and a larger indirect-dispatch memo (`kIndirectDispatchCacheEntries`),
a twelve-kart race start went from a 30 fps retrace lock to a steady 60.
The initial-exec TLS model is not an option: Bionic refuses it in a library
loaded with `dlopen`, which is how SDL loads the game
(`dlopen failed: TLS symbol ... using IE access model`).
- `runtime/cmake/PublicProducts.cmake` gains `MKW_GENERATED_DIR` so a build
configured from a checkout can name the translator output tree, and on
Android rewrites the PE/COFF `.section .rdata,"dr"` of a Windows-generated
+9 -1
View File
@@ -58,8 +58,11 @@ function(mkw_apply_common_compile_options target)
endfunction()
function(mkw_apply_translated_compile_options target)
# The NDK toolchain adds -fstack-protector-strong to every object. Translated guest code keeps
# its state in guest memory and the CpuContext, so the canaries only cost cycles there.
target_compile_options(${target} PRIVATE
-O2 ${MKW_TRANSLATED_PPC_FP_OPTIONS} -fno-slp-vectorize -w -pipe)
-O2 ${MKW_TRANSLATED_PPC_FP_OPTIONS} -fno-slp-vectorize -w -pipe
$<$<PLATFORM_ID:Android>:-fno-stack-protector>)
endfunction()
function(mkw_configure_object_target target)
@@ -319,6 +322,11 @@ function(mkw_configure_product target)
target_link_libraries(${target} PRIVATE mkw::libco ${CMAKE_DL_LIBS})
elseif(MKW_PLATFORM_ANDROID)
target_link_libraries(${target} PRIVATE mkw::libco android log vulkan ${CMAKE_DL_LIBS})
# A shared library's own calls and data references go through the PLT and GOT unless the
# symbols are bound at link time; with 29,000 translated functions calling each other and
# the runtime, those stubs were 4% of the game thread on the Quest. Nothing interposes
# symbols of the game library, and the JNI and SDL_main exports stay exported.
target_link_options(${target} PRIVATE "-Wl,-Bsymbolic")
endif()
if(MKW_PLATFORM_WINDOWS)
foreach(runtime_dll libc++.dll libunwind.dll)
+4 -1
View File
@@ -303,7 +303,10 @@ struct KnownTypedNativeCpuCall {
// a miss just re-runs the sorted lookup. 512 entries (12 KiB) covers the per-frame indirect
// working set while staying L1/L2 resident, unlike 4096 which would thrash L2. Must stay a
// power of two, the index masks with (size - 1).
inline constexpr size_t kIndirectDispatchCacheEntries = 512;
// Direct-mapped by a hash of the target address. A race keeps a few thousand distinct indirect
// targets live, so 512 entries missed about one call in three on the Quest and fell into the
// table's binary search; 8192 entries are 128 KB per memo and fit the big cores' L2.
inline constexpr size_t kIndirectDispatchCacheEntries = 8192;
// Namespace-scope `inline thread_local`, not function-local `static thread_local`, to avoid a
// thread-static init epoch check on every bctrl (same as g_currentCpuContext in ppc_runtime.h).