diff --git a/aurora-main/lib/gfx/common.cpp b/aurora-main/lib/gfx/common.cpp index 727ce3a..b8ffcad 100644 --- a/aurora-main/lib/gfx/common.cpp +++ b/aurora-main/lib/gfx/common.cpp @@ -120,6 +120,12 @@ static absl::flat_hash_map g_cachedSamplers; static ByteBuffer g_verts; static ByteBuffer g_uniforms; +// Interpolation reads uniforms again while matching and retaining a scene. +// Keep those reads in cached CPU memory, not a write-combined upload heap. +// The destination is latched per batch so live setting changes cannot switch +// its backing storage before pending interpolation tasks have completed. +static std::vector g_cpuUniforms; +static uint8_t* g_uniformUploadDestination = nullptr; static ByteBuffer g_indices; static ByteBuffer g_storage; static ByteBuffer g_textureUpload; @@ -1048,6 +1054,10 @@ void shutdown() { texture_replacement::shutdown(); gx::shutdown(); + g_uniformUploadDestination = nullptr; + g_uniforms.release(); + std::vector{}.swap(g_cpuUniforms); + g_textureUploads.clear(); g_cachedBindGroups.clear(); g_retiredBindGroups.clear(); @@ -1134,6 +1144,12 @@ static bool begin_frame_impl(bool clearEfb) { }; mapBuffer(g_verts, VertexBufferSize); mapBuffer(g_uniforms, UniformBufferSize); + g_uniformUploadDestination = nullptr; + if (gx::stereo_frame_interpolation_active()) { + g_cpuUniforms.resize(UniformBufferSize); + g_uniformUploadDestination = g_uniforms.data(); + g_uniforms = ByteBuffer{g_cpuUniforms.data(), g_cpuUniforms.size()}; + } mapBuffer(g_indices, IndexBufferSize); mapBuffer(g_storage, StorageBufferSize); if constexpr (UseTextureBuffer) { @@ -1179,6 +1195,7 @@ void abort_frame() noexcept { efb_ram::abort_async(); g_verts.release(); g_uniforms.release(); + g_uniformUploadDestination = nullptr; g_indices.release(); g_storage.release(); if constexpr (UseTextureBuffer) { @@ -1509,7 +1526,8 @@ static bool prepare_stereo_replay_uniforms(const StereoReplayFrame& stereoFrame, if (size == 0) return {}; const Range copy{static_cast(history->sources.size()), size}; - history->sources.insert(history->sources.end(), source, source + size); + history->sources.resize(history->sources.size() + size); + std::memcpy(history->sources.data() + copy.offset, source, size); return copy; }; saved->current = save(sourceUniform.data(), draw.uniformRange.size); @@ -1573,6 +1591,12 @@ static bool end_batch_impl(const wgpu::CommandEncoder& cmd, bool advanceFrame, bufferOffset += size; return writeSize; }; + if (g_uniformUploadDestination != nullptr) { + // Matching, endpoint capture and all uniform edits are complete. Upload + // only the used prefix, in one sequential write, before releasing the map. + std::memcpy(g_uniformUploadDestination, g_uniforms.data(), g_uniforms.size()); + g_uniformUploadDestination = nullptr; + } g_stagingBuffers[currentStagingBuffer].Unmap(); s_mappingState.store(BufferMapState::Unmapped, std::memory_order_release); g_stats.drawCallCount = g_drawCallCount; diff --git a/aurora-main/tests/stereo_frame_worker_smoke.cpp b/aurora-main/tests/stereo_frame_worker_smoke.cpp index 7daa77d..36204ec 100644 --- a/aurora-main/tests/stereo_frame_worker_smoke.cpp +++ b/aurora-main/tests/stereo_frame_worker_smoke.cpp @@ -49,7 +49,8 @@ int main(int argc, char** argv) { // Extra distinct draws expose CPU uniform/replay costs that a single triangle // cannot exercise. Keep the eye targets small to isolate that regression. const unsigned drawCount = argc > 1 ? std::max(1, std::atoi(argv[1])) : 1; - const bool interpolate = argc < 3 || std::atoi(argv[2]) != 0; + const int mode = argc < 3 ? 1 : std::clamp(std::atoi(argv[2]), 0, 2); + const bool indexed = argc > 3 && std::atoi(argv[3]) != 0; std::filesystem::create_directories("stereo-smoke-cache"); AuroraConfig config{}; config.appName = "Aurora VR interpolation smoke"; @@ -66,14 +67,15 @@ int main(int argc, char** argv) { config.xrInterop = true; aurora_initialize(argc, argv, &config); aurora_set_frame_interpolation_fps(0); - aurora_set_stereo_frame_interpolation(interpolate); + aurora_set_stereo_frame_interpolation(mode != 0); aurora_set_stereo_frame_provider(Provide, nullptr); aurora::stereo::set_sink(Encode, Submitted, nullptr); std::thread compositor([] { const auto start = Clock::now(); + uint64_t slot = 1; for (uint64_t token = 1;; ++token) { - const auto deadline = start + std::chrono::nanoseconds(token * 1'000'000'000 / 90); + const auto deadline = start + std::chrono::nanoseconds(slot * 1'000'000'000 / 90); std::this_thread::sleep_until(deadline); { std::lock_guard lock(packetMutex); @@ -100,6 +102,10 @@ int main(int argc, char** argv) { packetCv.wait(lock, [&] { return stop || completed == token; }); if (stop) return; + // A real compositor advances to a future display time after a missed + // tick. Do not burst old deadlines when interpolation is re-enabled. + const auto elapsed = std::chrono::duration_cast(Clock::now() - start).count(); + slot = std::max(slot + 1, static_cast(elapsed) * 90 / 1'000'000'000 + 1); } }); @@ -110,6 +116,10 @@ int main(int argc, char** argv) { aurora_update(); if (!aurora_begin_frame()) continue; + // A live setting change can arrive after this batch's backing memory was + // selected. Exercise both directions without restarting the renderer. + if (mode == 2 && frame % 60 == 0) + aurora_set_stereo_frame_interpolation((frame / 60) % 2 == 0); aurora_set_present_schedule( std::chrono::duration_cast(boundary.time_since_epoch()).count(), 16'666'667); Mtx44 projection{{1, 0, 0, 0}, {0, 1, 0, 0}, {0, 0, -1, -1}, {0, 0, -1, 0}}; @@ -119,6 +129,8 @@ int main(int argc, char** argv) { GXSetViewport(0, 0, 160, 120, 0, 1); GXSetScissor(0, 0, 160, 120); GXClearVtxDesc(); + if (indexed) + GXSetVtxDesc(GX_VA_PNMTXIDX, GX_DIRECT); GXSetVtxDesc(GX_VA_POS, GX_DIRECT); GXSetVtxAttrFmt(GX_VTXFMT0, GX_VA_POS, GX_POS_XYZ, GX_F32, 0); GXSetNumTexGens(0); @@ -126,13 +138,25 @@ int main(int argc, char** argv) { GXSetNumTevStages(1); GXSetTevOrder(GX_TEVSTAGE0, GX_TEXCOORD_NULL, GX_TEXMAP_NULL, GX_COLOR_NULL); GXSetTevOp(GX_TEVSTAGE0, GX_PASSCLR); + if (indexed) { + for (unsigned matrix = 0; matrix < 10; ++matrix) + GXLoadPosMtxImm(transform, matrix * 3); + } for (unsigned draw = 0; draw < drawCount; ++draw) { transform[1][3] = static_cast(draw % 20) * 0.01f; GXLoadPosMtxImm(transform, GX_PNMTX0); - GXBegin(GX_TRIANGLES, GX_VTXFMT0, 3); - GXPosition3f32(-1 + static_cast(draw) * 0.0001f, -1, 0); - GXPosition3f32(1, -1, 0); - GXPosition3f32(0, 1, 0); + GXBegin(GX_TRIANGLES, GX_VTXFMT0, indexed ? 30 : 3); + for (unsigned matrix = 0; matrix < (indexed ? 10u : 1u); ++matrix) { + if (indexed) + GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); + GXPosition3f32(-1 + static_cast(draw) * 0.0001f, -1, 0); + if (indexed) + GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); + GXPosition3f32(1, -1, 0); + if (indexed) + GXMatrixIndex1u8(GX_VA_PNMTXIDX, matrix * 3); + GXPosition3f32(0, 1, 0); + } GXEnd(); } aurora_end_frame_tagged(42); @@ -158,5 +182,6 @@ int main(int argc, char** argv) { submitted.load(), elapsed, fps); aurora_shutdown(); // More headset submissions must not come at the expense of simulation speed. - return 240 / elapsed > 55 && fps > (interpolate ? 85 : 55) && fps < (interpolate ? 100 : 65) ? 0 : 1; + const double target = mode == 2 ? 75 : mode == 1 ? 90 : 60; + return 240 / elapsed > 55 && fps > target - 5 && fps < target + 5 ? 0 : 1; }