diff --git a/aurora-main/lib/dolphin/gx/GXFrameBuffer.cpp b/aurora-main/lib/dolphin/gx/GXFrameBuffer.cpp index 3a6662e..e614633 100644 --- a/aurora-main/lib/dolphin/gx/GXFrameBuffer.cpp +++ b/aurora-main/lib/dolphin/gx/GXFrameBuffer.cpp @@ -15,8 +15,25 @@ #include #include #include +#include + +#if defined(__ANDROID__) +#include +#endif namespace { +#if defined(__ANDROID__) +bool quest1_direct_display_copy() noexcept { + static const bool enabled = [] { + char device[PROP_VALUE_MAX]{}; + return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0; + }(); + return enabled; +} +#else +constexpr bool quest1_direct_display_copy() noexcept { return false; } +#endif + struct CopyClearState { bool clearColor = false; bool clearAlpha = false; @@ -417,7 +434,15 @@ void GXCopyDisp(void* dest, GXBool clear) { const auto logicalDstHeight = std::max( g_gxState.dispCopyDstHeight != 0 ? g_gxState.dispCopyDstHeight : static_cast(g_gxState.dispCopySrc.height), 1); - const auto [dstWidth, dstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight); + auto [dstWidth, dstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight); + if (quest1_direct_display_copy()) { + // The Quest 1's Adreno 540 driver renders Aurora's filtered display + // copy as uniform black/white. Keeping the destination identical to the + // resolved EFB region selects WebGPU's plain CopyTextureToTexture path; + // the later presentation pass performs the final screen scaling. + dstWidth = static_cast(std::max(rect.width, 1)); + dstHeight = static_cast(std::max(rect.height, 1)); + } if (!g_gxState.displayCopyTexture || g_gxState.displayCopyWidth != dstWidth || g_gxState.displayCopyHeight != dstHeight) { @@ -431,6 +456,15 @@ void GXCopyDisp(void* dest, GXBool clear) { if (aurora::g_config.disableCopyFilter) { copyFilter = {0, copyFilter[0] + copyFilter[1] + copyFilter[2], 0}; } + if (quest1_direct_display_copy()) { + copyFilter = {0, 64, 0}; + static bool questDirectDisplayCopyLogged = false; + if (!questDirectDisplayCopyLogged) { + questDirectDisplayCopyLogged = true; + std::fprintf(stderr, "[gx] Quest 1 direct display copy: EFB rect %dx%d at %d,%d -> %ux%u\n", + rect.width, rect.height, rect.x, rect.y, dstWidth, dstHeight); + } + } aurora::gfx::resolve_pass(g_gxState.displayCopyTexture, rect, clearState.clearColor, clearState.clearAlpha, clearState.clearDepth, clearState.clearColorValue, aurora::gx::clear_depth_value(), GX_TF_RGBA8, nullptr, false, ©Filter, false, @@ -450,11 +484,21 @@ void GXCopyTex(void* dest, GXBool clear) { // Keep guest dimensions for cache identity while preserving scaled GPU detail. const auto logicalDstWidth = std::max(g_gxState.texCopyDstWidth, 1); const auto logicalDstHeight = std::max(g_gxState.texCopyDstHeight, 1); - const auto [scaledDstWidth, scaledDstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight); + auto [scaledDstWidth, scaledDstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight); const auto texCopyFmt = g_gxState.texCopyFmt; const bool sourceHasAlpha = aurora::gx::render_target_has_alpha(g_gxState.pixelFmt); const bool forceOpaqueAlpha = !sourceHasAlpha && !aurora::gx::is_depth_format(texCopyFmt); - const auto resolveFmt = texCopyFmt; + const bool quest1DirectRgb5a3 = quest1_direct_display_copy() && texCopyFmt == GX_TF_RGB5A3; + auto resolveFmt = texCopyFmt; + if (quest1DirectRgb5a3) { + // The host texture is RGBA8 regardless of the guest cache format. Match + // the EFB rectangle and advertise an RGBA8 resolve so this live menu copy + // becomes CopyTextureToTexture instead of invoking Adreno's broken + // RGB5A3 conversion/scaling shader. + scaledDstWidth = static_cast(std::max(rect.width, 1)); + scaledDstHeight = static_cast(std::max(rect.height, 1)); + resolveFmt = GX_TF_RGBA8; + } const aurora::gx::GXState::CopyTextureKey key{ .dest = dest, @@ -511,7 +555,10 @@ void GXCopyTex(void* dest, GXBool clear) { if (aurora::gx::render_target_has_alpha(g_gxState.pixelFmt)) { clearState.clearAlpha = clear && alphaUpdate; } - const auto copyFilter = combined_copy_filter_coefficients(g_gxState.copyFilterVFilter); + auto copyFilter = combined_copy_filter_coefficients(g_gxState.copyFilterVFilter); + if (quest1DirectRgb5a3) { + copyFilter = {0, 64, 0}; + } // Skip only recurring color copies so one-shot copies are never lost. const bool producedConsecutively = handle.revision != 0 && currentFrame - handle.lastProducedFrame <= 1; const bool persistentCopy = !aurora::gx::is_depth_format(texCopyFmt) && !producedConsecutively; @@ -524,7 +571,8 @@ void GXCopyTex(void* dest, GXBool clear) { } aurora::gfx::resolve_pass(handle.handle, rect, clearState.clearColor, clearState.clearAlpha, clearState.clearDepth, clearState.clearColorValue, aurora::gx::clear_depth_value(), resolveFmt, - &sourceRect.sampleRect, g_gxState.texCopyHalfScale, ©Filter, forceOpaqueAlpha, + quest1DirectRgb5a3 ? nullptr : &sourceRect.sampleRect, + quest1DirectRgb5a3 ? false : g_gxState.texCopyHalfScale, ©Filter, forceOpaqueAlpha, sourceRect.sampleRect.w() / std::max(g_gxState.texCopySrc.height, 1.0f), (g_gxState.copyClamp & GX_CLAMP_TOP) != 0, (g_gxState.copyClamp & GX_CLAMP_BOTTOM) != 0, persistentCopy); diff --git a/aurora-main/lib/gfx/pipeline_cache.cpp b/aurora-main/lib/gfx/pipeline_cache.cpp index 0038174..9b23d95 100644 --- a/aurora-main/lib/gfx/pipeline_cache.cpp +++ b/aurora-main/lib/gfx/pipeline_cache.cpp @@ -23,6 +23,11 @@ #include #include +#if defined(__ANDROID__) +#include +#include +#endif + namespace aurora::gfx { static Module Log("aurora::gfx::pipeline_cache"); @@ -65,9 +70,26 @@ constexpr size_t MaxQueuedPipelineBuilds = 256; // First-use compilation works best as a short parallel burst. Leave two logical processors for the // render and game threads, and cap large hosts to limit driver submissions and memory use. constexpr size_t ReservedLogicalProcessors = 2; +#if defined(__ANDROID__) +// Quest reports all eight Kryo cores, but the Adreno Vulkan driver serializes +// much of vkCreateGraphicsPipelines. Six equal-priority compiler threads crowd +// out the translated game and XR submission threads without materially +// shortening first-use compilation. Two background-priority workers keep +// pipeline discovery asynchronous while preserving frame cadence. +constexpr size_t MaxPipelineWorkers = 2; +#else constexpr size_t MaxPipelineWorkers = 22; -// Cached clear and GX pipelines are prewarmed using the full worker pool. +#endif +// Cached clear and GX pipelines are normally prewarmed using the full worker +// pool. Mobile Adreno drivers serialize much of pipeline creation internally; +// prewarming hundreds of recipes there starves first-use pipelines for over a +// minute. Keep cached recipes dormant on Android. A recipe is promoted to the +// priority queue as soon as the game actually requests it. +#if defined(__ANDROID__) +constexpr size_t MaxBackgroundPipelineWorkers = 0; +#else constexpr size_t MaxBackgroundPipelineWorkers = MaxPipelineWorkers; +#endif // For synchronous pipeline fallback (OpenGL) #ifdef NDEBUG constexpr size_t BuildPipelinesPerFrame = 5; @@ -462,6 +484,12 @@ static PipelineRef find_pipeline_impl(ShaderType type, const PipelineConfig& con } } else if (g_pendingPipelines.contains(hash)) { auto* pending = touch_pending_pipeline(hash, g_pipelineFrameActive); + // A cached Android recipe can sit dormant because background prewarm is + // disabled. Promoting it to first-use priority must wake a worker before + // bind_pipeline waits for completion, or both threads sleep forever. + if (g_pipelineFrameActive && deferGxPipeline) { + notifyWorker = true; + } if (pending != nullptr && firstFrameUsed < pending->firstFrameUsed) { pending->firstFrameUsed = firstFrameUsed; if (persist) { @@ -1033,6 +1061,12 @@ static void pipeline_worker() { #ifdef TRACY_ENABLE tracy::SetThreadName("Pipeline compilation thread"); #endif +#if defined(__ANDROID__) + pthread_setname_np(pthread_self(), "GXPipeline"); + // setpriority(PRIO_PROCESS, 0, ...) targets the calling Linux thread. + // Pipeline creation may finish later, but must not preempt gameplay or XR. + setpriority(PRIO_PROCESS, 0, 5); +#endif while (true) { PendingPipeline pending; diff --git a/aurora-main/lib/gfx/tex_copy_conv.cpp b/aurora-main/lib/gfx/tex_copy_conv.cpp index 6b68864..c513027 100644 --- a/aurora-main/lib/gfx/tex_copy_conv.cpp +++ b/aurora-main/lib/gfx/tex_copy_conv.cpp @@ -9,6 +9,12 @@ #include +#if defined(__ANDROID__) +#include +#endif + +#include + #include "texture_convert.hpp" using namespace std::string_literals; @@ -194,6 +200,48 @@ static constexpr std::string_view FragPassthrough = R"( } )"sv; +// The Quest 1's Adreno compiler misrenders the general copy shader, whose +// sample helper contains texture-dimension math and a dynamic filter branch, +// even for passthrough RGBA copies. This equivalent shader deliberately keeps +// the same bind-group contract while omitting that unused filter machinery. +static constexpr std::string_view SimpleBlitShader = R"( +@group(0) @binding(0) var src_samp: sampler; +@group(0) @binding(1) var src: texture_2d; + +struct UVTransform { + offset: vec2f, + scale: vec2f, + copy_filter: vec4f, + flags: vec4f, +}; +@group(0) @binding(2) var uv_xf: UVTransform; + +struct VertexOutput { + @builtin(position) pos: vec4f, + @location(0) uv: vec2f, +}; + +var positions: array = array( + vec2f(-1.0, 1.0), vec2f(-1.0, -3.0), vec2f(3.0, 1.0)); +var uvs: array = array( + vec2f(0.0, 0.0), vec2f(0.0, 2.0), vec2f(2.0, 0.0)); + +@vertex fn vs_main(@builtin(vertex_index) vi: u32) -> VertexOutput { + var out: VertexOutput; + out.pos = vec4f(positions[vi], 0.0, 1.0); + out.uv = uvs[vi] * uv_xf.scale + uv_xf.offset; + return out; +} + +@fragment fn fs_main(in: VertexOutput) -> @location(0) vec4f { + let color = textureSample(src, src_samp, in.uv); + if (uv_xf.flags.x != 0.0) { + return vec4f(color.rgb, 1.0); + } + return color; +} +)"sv; + // GX_TF_I4: 4-bit intensity -> R8Unorm (quantized) static constexpr std::string_view FragI4 = R"( @fragment fn fs_main(in: VertexOutput) -> @location(0) vec4f { @@ -374,6 +422,19 @@ static wgpu::Sampler g_nearestSampler; static wgpu::Sampler g_linearSampler; static absl::flat_hash_map g_pipelines; static wgpu::RenderPipeline g_blitPipeline; +static wgpu::RenderPipeline g_quest1BlitPipeline; + +static bool quest1_simple_blit() noexcept { +#if defined(__ANDROID__) + static const bool enabled = [] { + char device[PROP_VALUE_MAX]{}; + return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0; + }(); + return enabled; +#else + return false; +#endif +} static wgpu::RenderPipeline create_pipeline(const ConvPipeline& conv, const std::string_view shaderPreamble, const wgpu::BindGroupLayout& bindGroupLayout) { @@ -490,6 +551,9 @@ void initialize() { g_blitPipeline = create_pipeline( {GX_TF_RGBA8, FragPassthrough, webgpu::g_graphicsConfig.surfaceConfiguration.format, "TexCopyConv Blit"}, ShaderPreamble, g_bindGroupLayout); + g_quest1BlitPipeline = create_pipeline( + {GX_TF_RGBA8, {}, webgpu::g_graphicsConfig.surfaceConfiguration.format, "Quest 1 Simple TexCopy Blit"}, + SimpleBlitShader, g_bindGroupLayout); for (const auto& conv : ConvPipelines) { g_pipelines[conv.fmt] = create_pipeline(conv, ShaderPreamble, g_bindGroupLayout); if (conv.outputFormat != to_wgpu(conv.fmt)) { @@ -521,6 +585,7 @@ void initialize() { void shutdown() { g_pipelines.clear(); g_blitPipeline = {}; + g_quest1BlitPipeline = {}; g_bindGroupLayout = {}; g_depthBindGroupLayout = {}; g_nearestSampler = {}; @@ -595,6 +660,14 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con } void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { + if (quest1_simple_blit() && req.fmt == GX_TF_RGB5A3) { + // MKW's character/kart menu previews are live RGB5A3 EFB copies. The + // Adreno 540 miscompiles their quantizing conversion shader just like the + // filtered display-copy shader, leaving the preview panels black. Preserve + // the RGBA source directly; only the Wii-era RGB5A3 quantization is lost. + execute(cmd, req, g_quest1BlitPipeline); + return; + } const auto it = g_pipelines.find(req.fmt); if (it == g_pipelines.end()) { Log.fatal("No copy conversion pipeline for format {}", static_cast(req.fmt)); @@ -602,6 +675,8 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { execute(cmd, req, it->second); } -void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { execute(cmd, req, g_blitPipeline); } +void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { + execute(cmd, req, quest1_simple_blit() ? g_quest1BlitPipeline : g_blitPipeline); +} } // namespace aurora::gfx::tex_copy_conv diff --git a/docs/quest-port.md b/docs/quest-port.md index fbde6a3..fb65177 100644 --- a/docs/quest-port.md +++ b/docs/quest-port.md @@ -97,6 +97,29 @@ hit on Vulkan when a game's device was destroyed under the compositor. (there is no BGRA `AHardwareBuffer` format) and requests the two Dawn features the bridge needs. +### Quest 1 renderer compatibility + +Quest 1 identifies itself as Android device `monterey` and uses an Adreno 540 +driver that misrenders Aurora's general filtered EFB-copy shader. On that +device, `GXCopyDisp` keeps the destination equal to the resolved EFB region so +WebGPU can use `CopyTextureToTexture`; the later presentation pass still scales +the result. Live RGB5A3 menu copies use a minimal RGBA passthrough shader for +the same reason. Other Android devices retain the normal filtered and +format-converting paths. + +The workaround was isolated on a physical Quest 1: it changed the boot/menu +output from flickering black or white frames to a complete menu with character +and vehicle previews. It does intentionally skip Wii-era RGB5A3 quantization on +that device. + +Pipeline compilation is also scheduled differently on Android. Adreno +serializes much of `vkCreateGraphicsPipelines`, so a large equal-priority worker +pool starved the translated game and OpenXR pacing threads without shortening +the compile materially. Android uses two nice-level 5 workers and does not +prewarm cached recipes in the background; a cached recipe is promoted and its +workers are awakened when the game first requests it. Desktop keeps the full +worker pool and prewarm behavior. + ### Controllers Quest Touch controllers are not HID gamepads, so `openxr_input.cpp` syncs an