Merge PR #2 from darthcircuit: Quest 1 support

Adds the modernQuest and quest1 headset flavours (Kryo CPU target, direct-VR
library entry on Quest 1), records the CPU target in the game kit and checks
it wherever a kit or game package is used, and gates the Quest 1 EFB-copy and
pipeline-scheduling workarounds on the monterey device.

Resolved docs/quest-port.md by keeping both sides: the foveated rendering
section and the Quest 1 renderer compatibility section, and both Build-Quest.ps1
command lines.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
iChris4andClaude Opus 5.5 committed 2026-09-26 20:31:49 +02:00
commit 83805a410e
21 files changed
+431 -55

No files matched your search

+53 -5
View File
@@ -15,8 +15,25 @@
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#if defined(__ANDROID__)
#include <sys/system_properties.h>
#endif
namespace {
#if defined(__ANDROID__)
bool quest1_direct_display_copy() noexcept {
static const bool enabled = [] {
char device[PROP_VALUE_MAX]{};
return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0;
}();
return enabled;
}
#else
constexpr bool quest1_direct_display_copy() noexcept { return false; }
#endif
struct CopyClearState {
bool clearColor = false;
bool clearAlpha = false;
@@ -417,7 +434,15 @@ void GXCopyDisp(void* dest, GXBool clear) {
const auto logicalDstHeight = std::max<u32>(
g_gxState.dispCopyDstHeight != 0 ? g_gxState.dispCopyDstHeight : static_cast<u32>(g_gxState.dispCopySrc.height),
1);
const auto [dstWidth, dstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
auto [dstWidth, dstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
if (quest1_direct_display_copy()) {
// The Quest 1's Adreno 540 driver renders Aurora's filtered display
// copy as uniform black/white. Keeping the destination identical to the
// resolved EFB region selects WebGPU's plain CopyTextureToTexture path;
// the later presentation pass performs the final screen scaling.
dstWidth = static_cast<u32>(std::max(rect.width, 1));
dstHeight = static_cast<u32>(std::max(rect.height, 1));
}
if (!g_gxState.displayCopyTexture || g_gxState.displayCopyWidth != dstWidth ||
g_gxState.displayCopyHeight != dstHeight) {
@@ -431,6 +456,15 @@ void GXCopyDisp(void* dest, GXBool clear) {
if (aurora::g_config.disableCopyFilter) {
copyFilter = {0, copyFilter[0] + copyFilter[1] + copyFilter[2], 0};
}
if (quest1_direct_display_copy()) {
copyFilter = {0, 64, 0};
static bool questDirectDisplayCopyLogged = false;
if (!questDirectDisplayCopyLogged) {
questDirectDisplayCopyLogged = true;
std::fprintf(stderr, "[gx] Quest 1 direct display copy: EFB rect %dx%d at %d,%d -> %ux%u\n",
rect.width, rect.height, rect.x, rect.y, dstWidth, dstHeight);
}
}
aurora::gfx::resolve_pass(g_gxState.displayCopyTexture, rect, clearState.clearColor, clearState.clearAlpha,
clearState.clearDepth, clearState.clearColorValue, aurora::gx::clear_depth_value(),
GX_TF_RGBA8, nullptr, false, &copyFilter, false,
@@ -450,11 +484,21 @@ void GXCopyTex(void* dest, GXBool clear) {
// Keep guest dimensions for cache identity while preserving scaled GPU detail.
const auto logicalDstWidth = std::max<u32>(g_gxState.texCopyDstWidth, 1);
const auto logicalDstHeight = std::max<u32>(g_gxState.texCopyDstHeight, 1);
const auto [scaledDstWidth, scaledDstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
auto [scaledDstWidth, scaledDstHeight] = scale_copy_dst(logicalDstWidth, logicalDstHeight);
const auto texCopyFmt = g_gxState.texCopyFmt;
const bool sourceHasAlpha = aurora::gx::render_target_has_alpha(g_gxState.pixelFmt);
const bool forceOpaqueAlpha = !sourceHasAlpha && !aurora::gx::is_depth_format(texCopyFmt);
const auto resolveFmt = texCopyFmt;
const bool quest1DirectRgb5a3 = quest1_direct_display_copy() && texCopyFmt == GX_TF_RGB5A3;
auto resolveFmt = texCopyFmt;
if (quest1DirectRgb5a3) {
// The host texture is RGBA8 regardless of the guest cache format. Match
// the EFB rectangle and advertise an RGBA8 resolve so this live menu copy
// becomes CopyTextureToTexture instead of invoking Adreno's broken
// RGB5A3 conversion/scaling shader.
scaledDstWidth = static_cast<u32>(std::max(rect.width, 1));
scaledDstHeight = static_cast<u32>(std::max(rect.height, 1));
resolveFmt = GX_TF_RGBA8;
}
const aurora::gx::GXState::CopyTextureKey key{
.dest = dest,
@@ -518,7 +562,10 @@ void GXCopyTex(void* dest, GXBool clear) {
if (aurora::gx::render_target_has_alpha(g_gxState.pixelFmt)) {
clearState.clearAlpha = clear && alphaUpdate;
}
const auto copyFilter = combined_copy_filter_coefficients(g_gxState.copyFilterVFilter);
auto copyFilter = combined_copy_filter_coefficients(g_gxState.copyFilterVFilter);
if (quest1DirectRgb5a3) {
copyFilter = {0, 64, 0};
}
// Skip only recurring color copies so one-shot copies are never lost.
const bool producedConsecutively = handle.revision != 0 && currentFrame - handle.lastProducedFrame <= 1;
const bool persistentCopy = !aurora::gx::is_depth_format(texCopyFmt) && !producedConsecutively;
@@ -531,7 +578,8 @@ void GXCopyTex(void* dest, GXBool clear) {
}
aurora::gfx::resolve_pass(handle.handle, rect, clearState.clearColor, clearState.clearAlpha, clearState.clearDepth,
clearState.clearColorValue, aurora::gx::clear_depth_value(), resolveFmt,
&sourceRect.sampleRect, g_gxState.texCopyHalfScale, &copyFilter, forceOpaqueAlpha,
quest1DirectRgb5a3 ? nullptr : &sourceRect.sampleRect,
quest1DirectRgb5a3 ? false : g_gxState.texCopyHalfScale, &copyFilter, forceOpaqueAlpha,
sourceRect.sampleRect.w() / std::max<float>(g_gxState.texCopySrc.height, 1.0f),
(g_gxState.copyClamp & GX_CLAMP_TOP) != 0, (g_gxState.copyClamp & GX_CLAMP_BOTTOM) != 0,
persistentCopy);
+46 -5
View File
@@ -10,6 +10,7 @@
#include <atomic>
#include <chrono>
#include <condition_variable>
#include <cstring>
#include <deque>
#include <filesystem>
#include <limits>
@@ -23,6 +24,12 @@
#include <fmt/format.h>
#include <tracy/Tracy.hpp>
#if defined(__ANDROID__)
#include <pthread.h>
#include <sys/resource.h>
#include <sys/system_properties.h>
#endif
namespace aurora::gfx {
static Module Log("aurora::gfx::pipeline_cache");
@@ -66,8 +73,26 @@ constexpr size_t MaxQueuedPipelineBuilds = 256;
// render and game threads, and cap large hosts to limit driver submissions and memory use.
constexpr size_t ReservedLogicalProcessors = 2;
constexpr size_t MaxPipelineWorkers = 22;
// Cached clear and GX pipelines are prewarmed using the full worker pool.
constexpr size_t MaxBackgroundPipelineWorkers = MaxPipelineWorkers;
constexpr size_t Quest1MaxPipelineWorkers = 2;
static bool g_quest1PipelineScheduling = false;
static bool quest1_pipeline_scheduling() noexcept {
#if defined(__ANDROID__)
static const bool enabled = [] {
char device[PROP_VALUE_MAX]{};
return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0;
}();
return enabled;
#else
return false;
#endif
}
static size_t max_background_pipeline_workers() noexcept {
// Quest 1's Adreno driver serializes pipeline creation. Leave cached recipes
// dormant until first use there; all newer headsets retain normal prewarm.
return g_quest1PipelineScheduling ? 0 : MaxPipelineWorkers;
}
// For synchronous pipeline fallback (OpenGL)
#ifdef NDEBUG
constexpr size_t BuildPipelinesPerFrame = 5;
@@ -462,6 +487,12 @@ static PipelineRef find_pipeline_impl(ShaderType type, const PipelineConfig& con
}
} else if (g_pendingPipelines.contains(hash)) {
auto* pending = touch_pending_pipeline(hash, g_pipelineFrameActive);
// A cached recipe can sit dormant when background prewarm is disabled.
// Promoting it to first-use priority must wake a worker before
// bind_pipeline waits for completion, or both threads sleep forever.
if (g_pipelineFrameActive && deferGxPipeline) {
notifyWorker = true;
}
if (pending != nullptr && firstFrameUsed < pending->firstFrameUsed) {
pending->firstFrameUsed = firstFrameUsed;
if (persist) {
@@ -1033,6 +1064,14 @@ static void pipeline_worker() {
#ifdef TRACY_ENABLE
tracy::SetThreadName("Pipeline compilation thread");
#endif
#if defined(__ANDROID__)
pthread_setname_np(pthread_self(), "GXPipeline");
if (g_quest1PipelineScheduling) {
// setpriority(PRIO_PROCESS, 0, ...) targets the calling Linux thread.
// Quest 1 pipeline creation must not preempt gameplay or XR.
setpriority(PRIO_PROCESS, 0, 5);
}
#endif
while (true) {
PendingPipeline pending;
@@ -1042,7 +1081,7 @@ static void pipeline_worker() {
g_pipelineCv.wait(lock, [] {
return !g_priorityPipelines.empty() ||
(!g_backgroundPipelines.empty() &&
g_activeBackgroundPipelineWorkers < MaxBackgroundPipelineWorkers) ||
g_activeBackgroundPipelineWorkers < max_background_pipeline_workers()) ||
g_pipelineThreadEnd;
});
if (g_pipelineThreadEnd) {
@@ -1091,7 +1130,8 @@ static size_t pipeline_worker_count() {
}
const size_t availableWorkers =
logicalProcessors > ReservedLogicalProcessors ? logicalProcessors - ReservedLogicalProcessors : 1;
return std::clamp(availableWorkers, size_t{1}, MaxPipelineWorkers);
const size_t maximum = g_quest1PipelineScheduling ? Quest1MaxPipelineWorkers : MaxPipelineWorkers;
return std::clamp(availableWorkers, size_t{1}, maximum);
}
template <typename PipelineConfig, typename CreateFn>
@@ -1318,6 +1358,7 @@ void initialize_pipeline_cache() {
g_pipelineFrameActive = false;
g_pipelineThreadEnd = false;
g_activeBackgroundPipelineWorkers = 0;
g_quest1PipelineScheduling = quest1_pipeline_scheduling();
if (webgpu::g_backendType == wgpu::BackendType::OpenGL || webgpu::g_backendType == wgpu::BackendType::OpenGLES ||
webgpu::g_backendType == wgpu::BackendType::WebGPU) {
@@ -1330,7 +1371,7 @@ void initialize_pipeline_cache() {
g_pipelineThreads.emplace_back(pipeline_worker);
}
Log.info("Enabled {} priority pipeline compilation workers ({} background prewarm)",
workerCount, MaxBackgroundPipelineWorkers);
workerCount, max_background_pipeline_workers());
}
load_pipeline_cache();
+76 -1
View File
@@ -10,6 +10,12 @@
#include <absl/container/flat_hash_map.h>
#if defined(__ANDROID__)
#include <sys/system_properties.h>
#endif
#include <cstring>
#include "texture_convert.hpp"
using namespace std::string_literals;
@@ -195,6 +201,48 @@ static constexpr std::string_view FragPassthrough = R"(
}
)"sv;
// The Quest 1's Adreno compiler misrenders the general copy shader, whose
// sample helper contains texture-dimension math and a dynamic filter branch,
// even for passthrough RGBA copies. This equivalent shader deliberately keeps
// the same bind-group contract while omitting that unused filter machinery.
static constexpr std::string_view SimpleBlitShader = R"(
@group(0) @binding(0) var src_samp: sampler;
@group(0) @binding(1) var src: texture_2d<f32>;
struct UVTransform {
offset: vec2f,
scale: vec2f,
copy_filter: vec4f,
flags: vec4f,
};
@group(0) @binding(2) var<uniform> uv_xf: UVTransform;
struct VertexOutput {
@builtin(position) pos: vec4f,
@location(0) uv: vec2f,
};
var<private> positions: array<vec2f, 3> = array(
vec2f(-1.0, 1.0), vec2f(-1.0, -3.0), vec2f(3.0, 1.0));
var<private> uvs: array<vec2f, 3> = array(
vec2f(0.0, 0.0), vec2f(0.0, 2.0), vec2f(2.0, 0.0));
@vertex fn vs_main(@builtin(vertex_index) vi: u32) -> VertexOutput {
var out: VertexOutput;
out.pos = vec4f(positions[vi], 0.0, 1.0);
out.uv = uvs[vi] * uv_xf.scale + uv_xf.offset;
return out;
}
@fragment fn fs_main(in: VertexOutput) -> @location(0) vec4f {
let color = textureSample(src, src_samp, in.uv);
if (uv_xf.flags.x != 0.0) {
return vec4f(color.rgb, 1.0);
}
return color;
}
)"sv;
// GX_TF_I4: 4-bit intensity -> R8Unorm (quantized)
static constexpr std::string_view FragI4 = R"(
@fragment fn fs_main(in: VertexOutput) -> @location(0) vec4f {
@@ -375,6 +423,19 @@ static wgpu::Sampler g_nearestSampler;
static wgpu::Sampler g_linearSampler;
static absl::flat_hash_map<GXTexFmt, wgpu::RenderPipeline> g_pipelines;
static wgpu::RenderPipeline g_blitPipeline;
static wgpu::RenderPipeline g_quest1BlitPipeline;
static bool quest1_simple_blit() noexcept {
#if defined(__ANDROID__)
static const bool enabled = [] {
char device[PROP_VALUE_MAX]{};
return __system_property_get("ro.product.device", device) > 0 && std::strcmp(device, "monterey") == 0;
}();
return enabled;
#else
return false;
#endif
}
static wgpu::RenderPipeline create_pipeline(const ConvPipeline& conv, const std::string_view shaderPreamble,
const wgpu::BindGroupLayout& bindGroupLayout) {
@@ -491,6 +552,9 @@ void initialize() {
g_blitPipeline = create_pipeline(
{GX_TF_RGBA8, FragPassthrough, webgpu::g_graphicsConfig.surfaceConfiguration.format, "TexCopyConv Blit"},
ShaderPreamble, g_bindGroupLayout);
g_quest1BlitPipeline = create_pipeline(
{GX_TF_RGBA8, {}, webgpu::g_graphicsConfig.surfaceConfiguration.format, "Quest 1 Simple TexCopy Blit"},
SimpleBlitShader, g_bindGroupLayout);
for (const auto& conv : ConvPipelines) {
g_pipelines[conv.fmt] = create_pipeline(conv, ShaderPreamble, g_bindGroupLayout);
if (conv.outputFormat != to_wgpu(conv.fmt)) {
@@ -522,6 +586,7 @@ void initialize() {
void shutdown() {
g_pipelines.clear();
g_blitPipeline = {};
g_quest1BlitPipeline = {};
g_bindGroupLayout = {};
g_depthBindGroupLayout = {};
g_nearestSampler = {};
@@ -597,6 +662,14 @@ static void execute(const wgpu::CommandEncoder& cmd, const ConvRequest& req, con
}
void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
if (quest1_simple_blit() && req.fmt == GX_TF_RGB5A3) {
// MKW's character/kart menu previews are live RGB5A3 EFB copies. The
// Adreno 540 miscompiles their quantizing conversion shader just like the
// filtered display-copy shader, leaving the preview panels black. Preserve
// the RGBA source directly; only the Wii-era RGB5A3 quantization is lost.
execute(cmd, req, g_quest1BlitPipeline);
return;
}
const auto it = g_pipelines.find(req.fmt);
if (it == g_pipelines.end()) {
Log.fatal("No copy conversion pipeline for format {}", static_cast<int>(req.fmt));
@@ -604,6 +677,8 @@ void run(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
execute(cmd, req, it->second);
}
void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) { execute(cmd, req, g_blitPipeline); }
void blit(const wgpu::CommandEncoder& cmd, const ConvRequest& req) {
execute(cmd, req, quest1_simple_blit() ? g_quest1BlitPipeline : g_blitPipeline);
}
} // namespace aurora::gfx::tex_copy_conv