Quest: port device compatibility and pipeline fixes

This commit is contained in:
darthcircuit committed 2026-09-22 11:14:31 -06:00
1 parent f0e5e43985
commit 83fe04d160
4 files changed
+187 -7

No files matched your search

+53 -5
View File
@@ -15,8 +15,25 @@
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#if defined(__ANDROID__)
#include <sys/system_properties.h>
#endif
namespace {
#if defined(__ANDROID__)
bool quest1_direct_display_copy() noexcept {
static const bool enabled = [] {
char device[PROP_VALUE_MAX]{};
return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0;
}();
return enabled;
}
#else
constexpr bool quest1_direct_display_copy() noexcept { return false; }
#endif
struct CopyClearState {
bool clearColor = false;
bool clearAlpha = false;
@@ -417,7 +434,15 @@ void GXCopyDisp(void* dest, GXBool clear) {
const auto logicalDstHeight = std::max<u32>(
g_gxState.dispCopyDstHeight != 0 ? g_gxState.dispCopyDstHeight : static_cast<u32>(g_gxState.dispCopySrc.height),
1);
const auto [dstWidth, dstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
auto [dstWidth, dstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
if (quest1_direct_display_copy()) {
// The Quest 1's Adreno 540 driver renders Aurora's filtered display
// copy as uniform black/white. Keeping the destination identical to the
// resolved EFB region selects WebGPU's plain CopyTextureToTexture path;
// the later presentation pass performs the final screen scaling.
dstWidth = static_cast<u32>(std::max(rect.width, 1));
dstHeight = static_cast<u32>(std::max(rect.height, 1));
}
if (!g_gxState.displayCopyTexture || g_gxState.displayCopyWidth != dstWidth ||
g_gxState.displayCopyHeight != dstHeight) {
@@ -431,6 +456,15 @@ void GXCopyDisp(void* dest, GXBool clear) {
if (aurora::g_config.disableCopyFilter) {
copyFilter = {0, copyFilter[0] + copyFilter[1] + copyFilter[2], 0};
}
if (quest1_direct_display_copy()) {
copyFilter = {0, 64, 0};
static bool questDirectDisplayCopyLogged = false;
if (!questDirectDisplayCopyLogged) {
questDirectDisplayCopyLogged = true;
std::fprintf(stderr, "[gx] Quest 1 direct display copy: EFB rect %dx%d at %d,%d -> %ux%u\n",
rect.width, rect.height, rect.x, rect.y, dstWidth, dstHeight);
}
}
aurora::gfx::resolve_pass(g_gxState.displayCopyTexture, rect, clearState.clearColor, clearState.clearAlpha,
clearState.clearDepth, clearState.clearColorValue, aurora::gx::clear_depth_value(),
GX_TF_RGBA8, nullptr, false, &copyFilter, false,
@@ -450,11 +484,21 @@ void GXCopyTex(void* dest, GXBool clear) {
// Keep guest dimensions for cache identity while preserving scaled GPU detail.
const auto logicalDstWidth = std::max<u32>(g_gxState.texCopyDstWidth, 1);
const auto logicalDstHeight = std::max<u32>(g_gxState.texCopyDstHeight, 1);
const auto [scaledDstWidth, scaledDstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
auto [scaledDstWidth, scaledDstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
const auto texCopyFmt = g_gxState.texCopyFmt;
const bool sourceHasAlpha = aurora::gx::render_target_has_alpha(g_gxState.pixelFmt);
const bool forceOpaqueAlpha = !sourceHasAlpha && !aurora::gx::is_depth_format(texCopyFmt);
const auto resolveFmt = texCopyFmt;
const bool quest1DirectRgb5a3 = quest1_direct_display_copy() && texCopyFmt == GX_TF_RGB5A3;
auto resolveFmt = texCopyFmt;
if (quest1DirectRgb5a3) {
// The host texture is RGBA8 regardless of the guest cache format. Match
// the EFB rectangle and advertise an RGBA8 resolve so this live menu copy
// becomes CopyTextureToTexture instead of invoking Adreno's broken
// RGB5A3 conversion/scaling shader.
scaledDstWidth = static_cast<u32>(std::max(rect.width, 1));
scaledDstHeight = static_cast<u32>(std::max(rect.height, 1));
resolveFmt = GX_TF_RGBA8;
}
const aurora::gx::GXState::CopyTextureKey key{
.dest = dest,
@@ -511,7 +555,10 @@ void GXCopyTex(void* dest, GXBool clear) {
if (aurora::gx::render_target_has_alpha(g_gxState.pixelFmt)) {
clearState.clearAlpha = clear && alphaUpdate;
}
const auto copyFilter = combined_copy_filter_coefficients(g_gxState.copyFilterVFilter);
auto copyFilter = combined_copy_filter_coefficients(g_gxState.copyFilterVFilter);
if (quest1DirectRgb5a3) {
copyFilter = {0, 64, 0};
}
// Skip only recurring color copies so one-shot copies are never lost.
const bool producedConsecutively = handle.revision != 0 && currentFrame - handle.lastProducedFrame <= 1;
const bool persistentCopy = !aurora::gx::is_depth_format(texCopyFmt) && !producedConsecutively;
@@ -524,7 +571,8 @@ void GXCopyTex(void* dest, GXBool clear) {
}
aurora::gfx::resolve_pass(handle.handle, rect, clearState.clearColor, clearState.clearAlpha, clearState.clearDepth,
clearState.clearColorValue, aurora::gx::clear_depth_value(), resolveFmt,
&sourceRect.sampleRect, g_gxState.texCopyHalfScale, &copyFilter, forceOpaqueAlpha,
quest1DirectRgb5a3 ? nullptr : &sourceRect.sampleRect,
quest1DirectRgb5a3 ? false : g_gxState.texCopyHalfScale, &copyFilter, forceOpaqueAlpha,
sourceRect.sampleRect.w() / std::max<float>(g_gxState.texCopySrc.height, 1.0f),
(g_gxState.copyClamp & GX_CLAMP_TOP) != 0, (g_gxState.copyClamp & GX_CLAMP_BOTTOM) != 0,
persistentCopy);
+35 -1
View File
@@ -23,6 +23,11 @@
#include <fmt/format.h>
#include <tracy/Tracy.hpp>
#if defined(__ANDROID__)
#include <pthread.h>
#include <sys/resource.h>
#endif
namespace aurora::gfx {
static Module Log("aurora::gfx::pipeline_cache");
@@ -65,9 +70,26 @@ constexpr size_t MaxQueuedPipelineBuilds = 256;
// First-use compilation works best as a short parallel burst. Leave two logical processors for the
// render and game threads, and cap large hosts to limit driver submissions and memory use.
constexpr size_t ReservedLogicalProcessors = 2;
#if defined(__ANDROID__)
// Quest reports all eight Kryo cores, but the Adreno Vulkan driver serializes
// much of vkCreateGraphicsPipelines. Six equal-priority compiler threads crowd
// out the translated game and XR submission threads without materially
// shortening first-use compilation. Two background-priority workers keep
// pipeline discovery asynchronous while preserving frame cadence.
constexpr size_t MaxPipelineWorkers = 2;
#else
constexpr size_t MaxPipelineWorkers = 22;
// Cached clear and GX pipelines are prewarmed using the full worker pool.
#endif
// Cached clear and GX pipelines are normally prewarmed using the full worker
// pool. Mobile Adreno drivers serialize much of pipeline creation internally;
// prewarming hundreds of recipes there starves first-use pipelines for over a
// minute. Keep cached recipes dormant on Android. A recipe is promoted to the
// priority queue as soon as the game actually requests it.
#if defined(__ANDROID__)
constexpr size_t MaxBackgroundPipelineWorkers = 0;
#else
constexpr size_t MaxBackgroundPipelineWorkers = MaxPipelineWorkers;
#endif
// For synchronous pipeline fallback (OpenGL)
#ifdef NDEBUG
constexpr size_t BuildPipelinesPerFrame = 5;
@@ -462,6 +484,12 @@ static PipelineRef find_pipeline_impl(ShaderType type, const PipelineConfig& con
}
} else if (g_pendingPipelines.contains(hash)) {
auto* pending = touch_pending_pipeline(hash, g_pipelineFrameActive);
// A cached Android recipe can sit dormant because background prewarm is
// disabled. Promoting it to first-use priority must wake a worker before
// bind_pipeline waits for completion, or both threads sleep forever.
if (g_pipelineFrameActive && deferGxPipeline) {
notifyWorker = true;
}
if (pending != nullptr && firstFrameUsed < pending->firstFrameUsed) {
pending->firstFrameUsed = firstFrameUsed;
if (persist) {
@@ -1033,6 +1061,12 @@ static void pipeline_worker() {
#ifdef TRACY_ENABLE
tracy::SetThreadName("Pipeline compilation thread");
#endif
#if defined(__ANDROID__)
pthread_setname_np(pthread_self(), "GXPipeline");
// setpriority(PRIO_PROCESS, 0, ...) targets the calling Linux thread.
// Pipeline creation may finish later, but must not preempt gameplay or XR.
setpriority(PRIO_PROCESS, 0, 5);
#endif
while (true) {
PendingPipeline pending;
+76 -1
View File
@@ -9,6 +9,12 @@
#include <absl/container/flat_hash_map.h>
#if defined(__ANDROID__)
#include <sys/system_properties.h>
#endif
#include <cstring>
#include "texture_convert.hpp"
using namespace std::string_literals;
@@ -194,6 +200,48 @@ static constexpr std::string_view FragPassthrough = R"(
}
)"sv;
// The Quest 1's Adreno compiler misrenders the general copy shader, whose
// sample helper contains texture-dimension math and a dynamic filter branch,
// even for passthrough RGBA copies. This equivalent shader deliberately keeps
// the same bind-group contract while omitting that unused filter machinery.
static constexpr std::string_view SimpleBlitShader = R"(
@group(0) @binding(0) var src_samp: sampler;
@group(0) @binding(1) var src: texture_2d<f32>;
struct UVTransform {
offset: vec2f,
scale: vec2f,
copy_filter: vec4f,
flags: vec4f,
};
@group(0) @binding(2) var<uniform> uv_xf: UVTransform;
struct VertexOutput {
@builtin(position) pos: vec4f,
@location(0) uv: vec2f,
};
var<private> positions: array<vec2f, 3> = array(
vec2f(-1.0, 1.0), vec2f(-1.0, -3.0), vec2f(3.0, 1.0));
var<private> uvs: array<vec2f, 3> = array(
vec2f(0.0, 0.0), vec2f(0.0, 2.0), vec2f(2.0, 0.0));
@vertex fn vs_main(@builtin(vertex_index) vi: u32) -> VertexOutput {
var out: VertexOutput;
out.pos = vec4f(positions[vi], 0.0, 1.0);
out.uv = uvs[vi] * uv_xf.scale + uv_xf.offset;
return out;
}
@fragment fn fs_main(in: VertexOutput) -> @location(0) vec4f {
let color = textureSample(src, src_samp, in.uv);
if (uv_xf.flags.x != 0.0) {
return vec4f(color.rgb, 1.0);
}
return color;
}
)"sv;
// GX_TF_I4: 4-bit intensity -> R8Unorm (quantized)
static constexpr std::string_view FragI4 = R"(
@fragment fn fs_main(in: VertexOutput) -> @location(0) vec4f {
@@ -374,6 +422,19 @@ static wgpu::Sampler g_nearestSampler;
static wgpu::Sampler g_linearSampler;
static absl::flat_hash_map<GXTexFmt, wgpu::RenderPipeline> g_pipelines;
static wgpu::RenderPipeline g_blitPipeline;
static wgpu::RenderPipeline g_quest1BlitPipeline;
static bool quest1_simple_blit() noexcept {
#if defined(__ANDROID__)
static const bool enabled = [] {
char device[PROP_VALUE_MAX]{};
return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0;
}();
return enabled;
#else
return false;
#endif
}
static wgpu::RenderPipeline create_pipeline(const ConvPipeline& conv, const std::string_view shaderPreamble,
const wgpu::BindGroupLayout& bindGroupLayout) {
@@ -490,6 +551,9 @@ void initialize() {
g_blitPipeline = create_pipeline(
{GX_TF_RGBA8, FragPassthrough, webgpu::g_graphicsConfig.surfaceConfiguration.format, "TexCopyConv Blit"},
ShaderPreamble, g_bindGroupLayout);
g_quest1BlitPipeline = create_pipeline(
{GX_TF_RGBA8, {}, webgpu::g_graphicsConfig.surfaceConfiguration.format, "Quest 1 Simple TexCopy Blit"},
SimpleBlitShader, g_bindGroupLayout);
for (const auto& conv : ConvPipelines) {
g_pipelines[conv.fmt] = create_pipeline(conv, ShaderPreamble, g_bindGroupLayout);
if (conv.outputFormat != to_wgpu(conv.fmt)) {
@@ -521,6 +585,7 @@ void initialize() {
void shutdown() {
g_pipelines.clear();
g_blitPipeline = {};
g_quest1BlitPipeline = {};
g_bindGroupLayout = {};
g_depthBindGroupLayout = {};
g_nearestSampler = {};
@@ -595,6 +660,14 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
}
void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
if (quest1_simple_blit() && req.fmt == GX_TF_RGB5A3) {
// MKW's character/kart menu previews are live RGB5A3 EFB copies. The
// Adreno 540 miscompiles their quantizing conversion shader just like the
// filtered display-copy shader, leaving the preview panels black. Preserve
// the RGBA source directly; only the Wii-era RGB5A3 quantization is lost.
execute(cmd, req, g_quest1BlitPipeline);
return;
}
const auto it = g_pipelines.find(req.fmt);
if (it == g_pipelines.end()) {
Log.fatal("No copy conversion pipeline for format {}", static_cast<int>(req.fmt));
@@ -602,6 +675,8 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
execute(cmd, req, it->second);
}
void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { execute(cmd, req, g_blitPipeline); }
void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
execute(cmd, req, quest1_simple_blit() ? g_quest1BlitPipeline : g_blitPipeline);
}
} // namespace aurora::gfx::tex_copy_conv
+23
View File
@@ -97,6 +97,29 @@ hit on Vulkan when a game's device was destroyed under the compositor.
(there is no BGRA `AHardwareBuffer` format) and requests the two Dawn features
the bridge needs.
### Quest 1 renderer compatibility
Quest 1 identifies itself as Android device `monterey` and uses an Adreno 540
driver that misrenders Aurora's general filtered EFB-copy shader. On that
device, `GXCopyDisp` keeps the destination equal to the resolved EFB region so
WebGPU can use `CopyTextureToTexture`; the later presentation pass still scales
the result. Live RGB5A3 menu copies use a minimal RGBA passthrough shader for
the same reason. Other Android devices retain the normal filtered and
format-converting paths.
The workaround was isolated on a physical Quest 1: it changed the boot/menu
output from flickering black or white frames to a complete menu with character
and vehicle previews. It does intentionally skip Wii-era RGB5A3 quantization on
that device.
Pipeline compilation is also scheduled differently on Android. Adreno
serializes much of `vkCreateGraphicsPipelines`, so a large equal-priority worker
pool starved the translated game and OpenXR pacing threads without shortening
the compile materially. Android uses two nice-level 5 workers and does not
prewarm cached recipes in the background; a cached recipe is promoted and its
workers are awakened when the game first requests it. Desktop keeps the full
worker pool and prewarm behavior.
### Controllers
Quest Touch controllers are not HID gamepads, so `openxr_input.cpp` syncs an