From c2cc093ed068c770c7ce6a2ef884954abd93e64c Mon Sep 17 00:00:00 2001 From: shenlei Date: Tue, 29 Sep 2026 17:59:23 +0900 Subject: [PATCH] native: perf telemetry, 2 frames in flight, 120 Hz pacing, GPU shadow map - --perf-log CSV + logcat telemetry (perf_log.h, EterBase/PerfCounters.h), incl. display refresh rate / app vsync rate, thermal, CPU clocks; tools/pull_perf_logs.sh + tools/perf_summary.py - Vulkan renderer: two frames in flight, in-frame texture uploads (no vkQueueWaitIdle), --fps-cap / --refresh-rate / --frames-in-flight - Android: ANativeWindow_setFrameRate, ADPF performance hint, appCategory=game (ColorOS no longer drops to 60 Hz when idle) - Character shadow map rendered on the GPU into an offscreen render target instead of CPU rasterization + per-frame 1 MB upload; --cpu-shadow keeps the old path. Device (120 Hz, 32 mobs): 75 -> ~115 fps, script thread 8.9 -> 3.4 ms. draw_capture format v10. Co-Authored-By: Claude Opus 5.5 --- .../app/src/main/AndroidManifest.xml | 2 + .../org/metin2port/client/MainActivity.java | 119 +- .../src/platform/EterBase/PerfCounters.h | 124 ++ .../src/platform/EterLib/RecordingDevice.cpp | 79 ++ .../src/platform/EterLib/RenderCommands3D.h | 14 + .../src/platform/ScriptLib/PythonBoot.cpp | 13 + extension/src/platform/ScriptLib/PythonBoot.h | 3 + .../UserInterface/PythonApplication.cpp | 40 + extension/src/port/ScriptLib/PythonUtils.cpp | 7 + extension/tests/port_login_flow_server.cpp | 16 +- native_render/android_perf.h | 140 +++ native_render/draw_capture.h | 14 +- native_render/main.cpp | 1030 +++++++++++++---- native_render/perf_log.h | 419 +++++++ tools/perf_summary.py | 113 ++ tools/pull_perf_logs.sh | 11 + 16 files changed, 1883 insertions(+), 261 deletions(-) create mode 100644 extension/src/platform/EterBase/PerfCounters.h create mode 100644 native_render/android_perf.h create mode 100644 native_render/perf_log.h create mode 100644 tools/perf_summary.py create mode 100755 tools/pull_perf_logs.sh diff --git a/android-native/app/src/main/AndroidManifest.xml b/android-native/app/src/main/AndroidManifest.xml index 55ee99cb..6602f07c 100644 --- a/android-native/app/src/main/AndroidManifest.xml +++ b/android-native/app/src/main/AndroidManifest.xml @@ -7,6 +7,8 @@ = Build.VERSION_CODES.R ? getDisplay() + : getWindowManager().getDefaultDisplay(); + if (display == null) + return 0; + final Display.Mode current = display.getMode(); + Display.Mode best = null; + for (Display.Mode mode : display.getSupportedModes()) { + if (mode.getPhysicalWidth() != current.getPhysicalWidth() || + mode.getPhysicalHeight() != current.getPhysicalHeight()) + continue; + if (best == null) { + best = mode; + continue; + } + final float rate = mode.getRefreshRate(); + final boolean better = wanted < 0 ? rate > best.getRefreshRate() + : Math.abs(rate - wanted) < Math.abs(best.getRefreshRate() - wanted); + if (better) + best = mode; + } + if (best == null) + return 0; + final WindowManager.LayoutParams params = getWindow().getAttributes(); + params.preferredDisplayModeId = best.getModeId(); + getWindow().setAttributes(params); + Log.i("Metin2Args", "display mode " + best.getModeId() + " " + best.getPhysicalWidth() + "x" + + best.getPhysicalHeight() + " @ " + best.getRefreshRate() + " Hz"); + return Math.round(best.getRefreshRate()); + } + + // The app-vsync rate, measured with a Choreographer callback. It differs from the panel's refresh rate + // when the system renders the app at a divisor of it (ColorOS drops a non-game app to 60 fps after + // 2.5 s without a touch while the panel stays at 120 Hz). Started by the first currentRenderRate(). + private volatile float renderRate = -1f; + private boolean renderRateStarted = false; + + // Called from native code (android_perf.h), off the UI thread. + public float currentRenderRate() { + synchronized (this) { + if (!renderRateStarted) { + renderRateStarted = true; + runOnUiThread(() -> Choreographer.getInstance().postFrameCallback(new Choreographer.FrameCallback() { + long windowStart = 0; + int frames = 0; + + @Override + public void doFrame(long frameTimeNanos) { + if (windowStart == 0) { + windowStart = frameTimeNanos; + } else if (++frames >= 30) { + renderRate = frames * 1e9f / (frameTimeNanos - windowStart); + windowStart = frameTimeNanos; + frames = 0; + } + Choreographer.getInstance().postFrameCallback(this); + } + })); + } + } + return renderRate; + } + + // Called from native code (android_perf.h): battery temperature in C, -1 if unknown. + public float batteryTemperature() { + final Intent battery = registerReceiver(null, new IntentFilter(Intent.ACTION_BATTERY_CHANGED)); + final int tenths = battery != null ? battery.getIntExtra(BatteryManager.EXTRA_TEMPERATURE, -10) : -10; + return tenths / 10f; + } + + // Called from native code (android_perf.h): the rate the display runs at now. + public float currentRefreshRate() { + final Display display = Build.VERSION.SDK_INT >= Build.VERSION_CODES.R ? getDisplay() + : getWindowManager().getDefaultDisplay(); + return display != null ? display.getRefreshRate() : -1f; } @Override @@ -89,6 +204,8 @@ public class MainActivity extends SDLActivity { // Test phase: FPS readout in the top-left corner. Drop this line for release builds. if (!args.contains("--show-fps")) args = args.trim() + " --show-fps"; + if (!args.contains("--refresh-rate") && chosenRefreshRate > 0) + args = args.trim() + " --refresh-rate " + chosenRefreshRate; return args.trim().split("\\s+"); } diff --git a/extension/src/platform/EterBase/PerfCounters.h b/extension/src/platform/EterBase/PerfCounters.h new file mode 100644 index 00000000..1a7c296c --- /dev/null +++ b/extension/src/platform/EterBase/PerfCounters.h @@ -0,0 +1,124 @@ +#pragma once +// PORT: performance telemetry (40250 has no counterpart). The script thread adds the wall time of the +// sections of CPythonApplication::Process() and of the Python callbacks here; the native host +// (native_render/perf_log.h) reads and clears the totals every sample window. The two threads never run +// at the same time (PythonBoot hands one baton between them), so plain fields are enough. +// +// Nothing is timed until MtPerf::Enabled() is set, so a build without --perf-log pays one branch per +// section. + +#include + +namespace MtPerf +{ +enum ESection +{ + SECTION_PROCESS, // the whole CPythonApplication::Process() + SECTION_NETWORK, // CPythonNetworkStream + CAccountConnector Process + SECTION_CAMERA, // __UpdateCamera, CResourceManager::Update, OnCameraUpdate, OnMouseUpdate + SECTION_UI_UPDATE, // OnUIUpdate (WindowManager Update -> Python OnUpdate) + SECTION_UPDATE_GAME, // app.UpdateGame, inside UI_UPDATE + SECTION_RENDER_BEGIN, // CCullingManager::Update + the command-list restarts + SECTION_UI_RENDER, // OnUIRender (Python OnRender) + OnMouseRender + SECTION_RENDER_GAME, // app.RenderGame, inside UI_RENDER + SECTION_PYTHON, // outermost C++ -> Python callbacks (includes UPDATE_GAME/RENDER_GAME) + SECTION_SHADOW_RASTER, // RecordingDevice's CPU rasterization of the shadow texture, inside RENDER_GAME + SECTION_COUNT, +}; + +struct SCounters +{ + double ms[SECTION_COUNT] = {}; + unsigned calls[SECTION_COUNT] = {}; + unsigned pyMissing = 0; // callbacks whose method does not exist + int actorCount = 0; // CPythonCharacterManager alive instances, last Process() + int scriptCpu = -1; // CPU the script thread last ran on (Linux/Android), else -1 +}; + +inline bool& Enabled() +{ + static bool s_bEnabled = false; + return s_bEnabled; +} + +inline SCounters& Counters() +{ + static SCounters s_kCounters; + return s_kCounters; +} + +inline int& PythonDepth() +{ + static int s_iDepth = 0; + return s_iDepth; +} + +class CScope +{ + public: + explicit CScope(ESection eSection) : m_eSection(eSection), m_bActive(Enabled()) + { + if (m_bActive) + m_kStart = std::chrono::steady_clock::now(); + } + ~CScope() + { + if (!m_bActive) + return; + SCounters& rkCounters = Counters(); + rkCounters.ms[m_eSection] += std::chrono::duration(std::chrono::steady_clock::now() - m_kStart).count(); + ++rkCounters.calls[m_eSection]; + } + + private: + ESection m_eSection; + bool m_bActive; + std::chrono::steady_clock::time_point m_kStart; +}; + +// Times only the outermost Python callback, so nested callbacks are not counted twice. +class CPythonScope +{ + public: + CPythonScope() : m_bActive(Enabled() && PythonDepth()++ == 0) + { + if (m_bActive) + m_kStart = std::chrono::steady_clock::now(); + } + ~CPythonScope() + { + if (!Enabled()) + return; + if (PythonDepth() > 0) + --PythonDepth(); + if (!m_bActive) + return; + SCounters& rkCounters = Counters(); + rkCounters.ms[SECTION_PYTHON] += std::chrono::duration(std::chrono::steady_clock::now() - m_kStart).count(); + ++rkCounters.calls[SECTION_PYTHON]; + } + + private: + bool m_bActive; + std::chrono::steady_clock::time_point m_kStart; +}; + +inline void CountPythonMissing() +{ + if (Enabled()) + ++Counters().pyMissing; +} + +// The totals since the last call, which clears them. +inline SCounters Take() +{ + SCounters kTaken = Counters(); + SCounters& rkCounters = Counters(); + const int iActors = rkCounters.actorCount; + const int iCpu = rkCounters.scriptCpu; + rkCounters = SCounters(); + rkCounters.actorCount = iActors; + rkCounters.scriptCpu = iCpu; + return kTaken; +} +} diff --git a/extension/src/platform/EterLib/RecordingDevice.cpp b/extension/src/platform/EterLib/RecordingDevice.cpp index 32f752e8..2bcac33b 100644 --- a/extension/src/platform/EterLib/RecordingDevice.cpp +++ b/extension/src/platform/EterLib/RecordingDevice.cpp @@ -2,6 +2,7 @@ #include "RenderCommands3D.h" #include "UIRenderCommands.h" #include "CpuBuffer.h" +#include "../EterBase/PerfCounters.h" #include #include @@ -15,6 +16,7 @@ namespace { std::mutex g_draws_mutex; std::vector g_draws; int g_native_terrain_override = -1; +bool g_gpu_render_targets = false; struct VertexLayout { @@ -434,7 +436,33 @@ public: auto* surface = static_cast(m_renderTarget); if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5) return S_OK; + if (g_gpu_render_targets) + { + // The whole target, as the CPU path below clears it. + Render3DDraw clear; + clear.render_target = surface->parent->id; + clear.target_width = surface->desc.Width; + clear.target_height = surface->desc.Height; + clear.clear_flags = flags & (D3DCLEAR_TARGET | D3DCLEAR_ZBUFFER); + clear.clear_color = color; + clear.clear_z = depth; + if (clear.clear_flags) + Render3DAdd(std::move(clear)); + return S_OK; + } const size_t pixels = size_t(surface->desc.Width) * surface->desc.Height; + if (const char* dump = std::getenv("MT_DUMP_RT"); dump && *dump && surface->parent->levels[0].bytes.size() >= pixels * 2) + { + // Debug: the previous frame's CPU shadow map, red channel, before it is cleared. + if (FILE* out = std::fopen((std::string(dump) + "_cpu.pgm").c_str(), "wb")) + { + std::fprintf(out, "P5\n%u %u\n255\n", unsigned(surface->desc.Width), unsigned(surface->desc.Height)); + const auto& bytes = surface->parent->levels[0].bytes; + for (size_t i = 0; i < pixels; ++i) + std::fputc(int(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31), out); + std::fclose(out); + } + } if (flags & D3DCLEAR_TARGET) { const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 | @@ -593,6 +621,7 @@ private: // the normal memory-texture path upload the result for terrain/object projection. void rasterize_shadow(const Render3DDraw& draw) { + MtPerf::CScope kPerf(MtPerf::SECTION_SHADOW_RASTER); auto* surface = static_cast(m_renderTarget); if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5 || draw.lines) return; @@ -1048,10 +1077,47 @@ private: } if (m_renderTarget == m_backBuffer) Render3DAdd(std::move(draw)); + else if (g_gpu_render_targets) + record_shadow(std::move(draw)); else rasterize_shadow(draw); } + // GPU counterpart of rasterize_shadow: the draw goes to the renderer's offscreen pass with the + // states rasterize_shadow implements -- a flat D3DRS_TEXTUREFACTOR fill, no culling, no texture, + // light, fog or blending, depth test LESSEQUAL with depth writes -- so both paths draw the same map. + void record_shadow(Render3DDraw draw) + { + auto* surface = static_cast(m_renderTarget); + if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5 || draw.lines || + draw.pretransformed || draw.indices.size() < 3) + return; + draw.render_target = surface->parent->id; + draw.target_width = surface->desc.Width; + draw.target_height = surface->desc.Height; + draw.ui_offset_x = 0; + draw.texture0.clear(); + draw.texture1.clear(); + draw.texture_factor = m_renderStates[D3DRS_TEXTUREFACTOR]; + draw.color_op[0] = D3DTOP_SELECTARG1; + draw.color_arg1[0] = D3DTA_TFACTOR; + draw.alpha_op[0] = D3DTOP_SELECTARG1; + draw.alpha_arg1[0] = D3DTA_TFACTOR; + draw.color_op[1] = D3DTOP_DISABLE; + draw.alpha_op[1] = D3DTOP_DISABLE; + draw.lighting = FALSE; + draw.light0 = false; + for (auto& light : draw.lights) light.type = 0; + draw.fog_enable = FALSE; + draw.alpha_blend = FALSE; + draw.alpha_test = FALSE; + draw.cull_mode = D3DCULL_NONE; + draw.z_enable = TRUE; + draw.z_write = TRUE; + draw.z_func = D3DCMP_LESSEQUAL; + Render3DAdd(std::move(draw)); + } + ULONG m_refs = 1; int m_width, m_height; D3DVIEWPORT8 m_viewport = {}; @@ -1089,6 +1155,11 @@ std::string MtCpuTextureNameFromHandle(const IDirect3DBaseTexture8* handle) std::lock_guard lock(g_cpu_tex_mutex); for (const auto& entry : g_cpu_textures) { + if (entry.second != handle || entry.second->levels.empty()) + continue; + const D3DSURFACE_DESC& desc = entry.second->levels[0].desc; + if (g_gpu_render_targets && (desc.Usage & D3DUSAGE_RENDERTARGET) && desc.Format == D3DFMT_R5G6B5) + return Render3DRenderTargetName(entry.first, desc.Width, desc.Height); if (entry.second == handle && !entry.second->levels.empty()) return "mem:cpu_" + std::to_string(entry.first) + "@" + std::to_string(entry.second->revision); } @@ -1168,6 +1239,14 @@ bool IsNativeTerrainRenderEnabled() return env_on; } +void SetGpuRenderTargetsEnabled(bool enabled) { g_gpu_render_targets = enabled; } +bool IsGpuRenderTargetsEnabled() { return g_gpu_render_targets; } + +std::string Render3DRenderTargetName(std::uint32_t id, std::uint32_t width, std::uint32_t height) +{ + return "rt:" + std::to_string(id) + ":" + std::to_string(width) + "x" + std::to_string(height); +} + void Render3DBeginFrame() { std::lock_guard lock(g_draws_mutex); diff --git a/extension/src/platform/EterLib/RenderCommands3D.h b/extension/src/platform/EterLib/RenderCommands3D.h index 5d809af2..f06e313c 100644 --- a/extension/src/platform/EterLib/RenderCommands3D.h +++ b/extension/src/platform/EterLib/RenderCommands3D.h @@ -29,6 +29,13 @@ struct Render3DDraw { std::uint32_t clear_color = 0; // 0xAARRGGBB float clear_z = 1; + // Non-zero: the draw (or clear) targets an offscreen render-target texture instead of the back + // buffer -- the 40250 character shadow map. The renderer draws these into the texture named + // Render3DRenderTargetName(render_target, ...) ahead of the back-buffer pass, in recorded order. + // viewport is then in the target's pixels. Only emitted while GPU render targets are enabled. + std::uint32_t render_target = 0; + std::uint32_t target_width = 0, target_height = 0; + // Stage 0/1 textures, named like UIRenderTextureName: the pack path of a file texture or // "mem:@"; empty when the stage has no texture. std::string texture0; @@ -122,6 +129,13 @@ bool LookupGpuSkinSubrange(const void* vertex_buffer_base, std::uint32_t stride, std::uint32_t lo_vertex, std::uint32_t hi_vertex, GpuSkinSubrangeView* out_view); +// GPU render targets: shadow-map draws are recorded with render_target set and sampled as +// "rt::x" instead of being rasterized on the CPU into a "mem:cpu_" texture. Off by +// default (the Godot consumer has no offscreen pass); the native Vulkan renderer turns it on. +void SetGpuRenderTargetsEnabled(bool enabled); +bool IsGpuRenderTargetsEnabled(); +std::string Render3DRenderTargetName(std::uint32_t id, std::uint32_t width, std::uint32_t height); + void Render3DBeginFrame(); void Render3DAdd(Render3DDraw draw); const std::vector& Render3DDraws(); diff --git a/extension/src/platform/ScriptLib/PythonBoot.cpp b/extension/src/platform/ScriptLib/PythonBoot.cpp index 8c4b83dd..f150bc15 100644 --- a/extension/src/platform/ScriptLib/PythonBoot.cpp +++ b/extension/src/platform/ScriptLib/PythonBoot.cpp @@ -50,6 +50,7 @@ #include "../EterBase/LogBox.h" #include "../UserInterface/ServerClock.h" +#include #include #include #include @@ -61,6 +62,9 @@ #include #include #endif +#if defined(__linux__) || defined(__ANDROID__) +#include +#endif namespace { @@ -185,6 +189,7 @@ struct ScriptFiber }; ScriptFiber g_fiber; thread_local bool t_on_fiber = false; +std::atomic g_script_thread_id{0}; // Caller holds the GIL. Returns with the GIL held again after the script has given the baton back. void hand_to_script() @@ -218,6 +223,9 @@ void* fiber_main(void*) g_fiber.cv.wait(lock, [] { return g_fiber.turn == ScriptFiber::Script; }); } t_on_fiber = true; +#if defined(__linux__) || defined(__ANDROID__) + g_script_thread_id = int(gettid()); +#endif PyGILState_STATE gil = PyGILState_Ensure(); std::string error; const bool ok = run_main_script_body(g_fiber.command_line.c_str(), &error); @@ -927,6 +935,11 @@ bool RunMainScript(const char* lpCmdLine, std::string* error) return true; } +int ScriptThreadId() +{ + return g_script_thread_id; +} + bool IsAppLooping() { return g_fiber.alive && g_fiber.looping && !g_fiber.finished; diff --git a/extension/src/platform/ScriptLib/PythonBoot.h b/extension/src/platform/ScriptLib/PythonBoot.h index c9b5273d..08cef2e5 100644 --- a/extension/src/platform/ScriptLib/PythonBoot.h +++ b/extension/src/platform/ScriptLib/PythonBoot.h @@ -40,6 +40,9 @@ bool CursorVisible(); // CPythonApplication::Process() while the host thread waits, so the two never run concurrently. bool RunMainScript(const char* lpCmdLine, std::string* error); bool IsAppLooping(); +// The script thread's kernel thread id (Linux/Android), 0 before it starts or elsewhere. The host adds +// it to its ADPF performance-hint session. +int ScriptThreadId(); bool AppFrame(); // Native CPythonBackground map selection, in the local coordinate frame of RenderGame. std::string CurrentMapName(); diff --git a/extension/src/platform/UserInterface/PythonApplication.cpp b/extension/src/platform/UserInterface/PythonApplication.cpp index 561c8699..254d352e 100644 --- a/extension/src/platform/UserInterface/PythonApplication.cpp +++ b/extension/src/platform/UserInterface/PythonApplication.cpp @@ -22,6 +22,11 @@ #include "UserInterface/PythonSystem.h" #include "../ScriptLib/PythonBoot.h" #include "ServerClock.h" +#include "../EterBase/PerfCounters.h" + +#if defined(__linux__) || defined(__ANDROID__) +#include +#endif // 2V0-d link surface; 2V0-f adds the application lifecycle (Create/Loop/Process/Exit) that // prototype.RunApp() drives. The rest of the members still wait for the game slices (2V1-2V3). @@ -196,6 +201,7 @@ void CPythonApplication::GetMousePosition(POINT* ppt) // PORT: the members are reached through their singletons (see the note at the top of the file). void CPythonApplication::RenderGame() { + MtPerf::CScope kPerf(MtPerf::SECTION_RENDER_GAME); float fAspect=UI::CWindowManager::Instance().GetAspect(); float fFarClip=CPythonBackground::Instance().GetFarClip(); @@ -253,6 +259,7 @@ void CPythonApplication::RenderGame() // calls it every frame. PORT: members through their singletons, as in RenderGame. void CPythonApplication::UpdateGame() { + MtPerf::CScope kPerf(MtPerf::SECTION_UPDATE_GAME); POINT ptMouse; GetMousePosition(&ptMouse); @@ -293,19 +300,42 @@ void CPythonApplication::UpdateGame() // input instead), camera (2V3), resource GC, mouse/UI update, then the render block's interface pass. // PORT: the frame-skip bookkeeping and the game render pass are left out; the Godot frame clock // decides when this runs, and RenderGame belongs to 2V3. +// PORT: telemetry for --perf-log (platform/EterBase/PerfCounters.h); 40250 has no counterpart. +static void __PerfSampleWorld() +{ + if (!MtPerf::Enabled()) + return; + MtPerf::SCounters& rkCounters = MtPerf::Counters(); + CPythonCharacterManager& rkChrMgr = CPythonCharacterManager::Instance(); + int iActors = 0; + for (CPythonCharacterManager::CharacterIterator i = rkChrMgr.CharacterInstanceBegin(); i != rkChrMgr.CharacterInstanceEnd(); ++i) + ++iActors; + rkCounters.actorCount = iActors; +#if defined(__linux__) || defined(__ANDROID__) + rkCounters.scriptCpu = sched_getcpu(); +#endif +} + bool CPythonApplication::Process() { + MtPerf::CScope kPerfProcess(MtPerf::SECTION_PROCESS); + __PerfSampleWorld(); ELTimer_SetFrameMSec(); CTimer& rkTimer=CTimer::Instance(); rkTimer.Advance(); // Network I/O + { + MtPerf::CScope kPerf(MtPerf::SECTION_NETWORK); CPythonNetworkStream::Instance().Process(); // PORT: m_kGuildMarkUploader/m_kGuildMarkDownloader.Process() follow here in 40250; the guild mark // transfer is not ported yet (GuildMarkDownloader.cpp/GuildMarkUploader.cpp). CAccountConnector::Instance().Process(); + } + { + MtPerf::CScope kPerf(MtPerf::SECTION_CAMERA); //!@# Alt+Tab 중 SetTransfor 에서 튕김 현상 해결을 위해 - [levites] //if (m_isActivateWnd) __UpdateCamera(); @@ -315,7 +345,11 @@ bool CPythonApplication::Process() OnCameraUpdate(); if (IsNativeTerrainRenderEnabled()) OnMouseUpdate(); + } + { + MtPerf::CScope kPerf(MtPerf::SECTION_UI_UPDATE); OnUIUpdate(); + } if (IsNativeTerrainRenderEnabled()) CGrannyMaterial::TranslateSpecularMatrix(g_specularSpd, g_specularSpd, 0.0f); @@ -325,14 +359,19 @@ bool CPythonApplication::Process() // Win32 activation (CMSWindow::IsActive is a stub). The frame's UI and 3D command lists restart here. // 40250 PythonApplication.cpp:681. DWORD dwRenderStartTime = ELTimer_GetMSec(); + { + MtPerf::CScope kPerf(MtPerf::SECTION_RENDER_BEGIN); CCullingManager::Instance().Update(); UIRenderBeginFrame(); Render3DBeginFrame(); + } CPythonGraphic& rkGraphic = CPythonGraphic::Instance(); if (rkGraphic.Begin()) { rkGraphic.SetInterfaceRenderState(); + { + MtPerf::CScope kPerf(MtPerf::SECTION_UI_RENDER); OnUIRender(); if (IsNativeTerrainRenderEnabled()) { @@ -342,6 +381,7 @@ bool CPythonApplication::Process() UIRenderSetClip(0.0f, 0.0f, float(cw), float(ch)); OnMouseRender(); } + } rkGraphic.End(); diff --git a/extension/src/port/ScriptLib/PythonUtils.cpp b/extension/src/port/ScriptLib/PythonUtils.cpp index 424b989a..ee9ee361 100644 --- a/extension/src/port/ScriptLib/PythonUtils.cpp +++ b/extension/src/port/ScriptLib/PythonUtils.cpp @@ -1,5 +1,6 @@ #include "StdAfx.h" #include "PythonUtils.h" +#include "../../platform/EterBase/PerfCounters.h" // PORT: --perf-log callback timing IPythonExceptionSender * g_pkExceptionSender = NULL; @@ -291,6 +292,7 @@ bool PyCallClassMemberFunc(PyObject* poClass, const char* c_szFunc, PyObject* po */ bool __PyCallClassMemberFunc_ByCString(PyObject* poClass, const char* c_szFunc, PyObject* poArgs, PyObject** ppoRet) { + MtPerf::CPythonScope kPerf; if (!poClass) { Py_XDECREF(poArgs); @@ -302,6 +304,7 @@ bool __PyCallClassMemberFunc_ByCString(PyObject* poClass, const char* c_szFunc, if (!poFunc) { PyErr_Clear(); + MtPerf::CountPythonMissing(); Py_XDECREF(poArgs); return false; } @@ -339,6 +342,7 @@ bool __PyCallClassMemberFunc_ByCString(PyObject* poClass, const char* c_szFunc, bool __PyCallClassMemberFunc_ByPyString(PyObject* poClass, PyObject* poFuncName, PyObject* poArgs, PyObject** ppoRet) { + MtPerf::CPythonScope kPerf; if (!poClass) { Py_XDECREF(poArgs); @@ -350,6 +354,7 @@ bool __PyCallClassMemberFunc_ByPyString(PyObject* poClass, PyObject* poFuncName, if (!poFunc) { PyErr_Clear(); + MtPerf::CountPythonMissing(); Py_XDECREF(poArgs); return false; } @@ -387,6 +392,7 @@ bool __PyCallClassMemberFunc_ByPyString(PyObject* poClass, PyObject* poFuncName, bool __PyCallClassMemberFunc(PyObject* poClass, PyObject * poFunc, PyObject* poArgs, PyObject** ppoRet) { + MtPerf::CPythonScope kPerf; if (!poClass) { Py_XDECREF(poArgs); @@ -396,6 +402,7 @@ bool __PyCallClassMemberFunc(PyObject* poClass, PyObject * poFunc, PyObject* poA if (!poFunc) { PyErr_Clear(); + MtPerf::CountPythonMissing(); Py_XDECREF(poArgs); return false; } diff --git a/extension/tests/port_login_flow_server.cpp b/extension/tests/port_login_flow_server.cpp index 8a9f9623..9531fda5 100644 --- a/extension/tests/port_login_flow_server.cpp +++ b/extension/tests/port_login_flow_server.cpp @@ -22,7 +22,17 @@ namespace { constexpr std::uint32_t kHandshake = 0x2468ace0; constexpr int kHandshakeRetryLimit = 32; // game/src/desc.h HANDSHAKE_RETRY_LIMIT -constexpr int kIdleMs = 20000; +// A client that sends nothing for this long is dropped. MT_FAKE_IDLE_MS raises it for runs that stand +// still in game (perf measurements). +int idle_ms() +{ + static const int ms = [] { + const char* value = std::getenv("MT_FAKE_IDLE_MS"); + const long parsed = value && *value ? std::strtol(value, nullptr, 10) : 0; + return parsed >= 1000 ? static_cast(parsed) : 20000; + }(); + return ms; +} int fake_mob_count() { @@ -127,7 +137,7 @@ struct FakeLoginServer::Connection int ready = poll(&pfd, 1, 50); if (ready == 0) { - if ((waited += 50) >= kIdleMs) + if ((waited += 50) >= idle_ms()) return -1; continue; } @@ -222,7 +232,7 @@ bool FakeLoginServer::Accept(int listener, Connection& c) c.fd = accept(listener, nullptr, nullptr); return c.fd >= 0; } - if (waited >= kIdleMs) + if (waited >= idle_ms()) return Fail("no client connected"); } return false; diff --git a/native_render/android_perf.h b/native_render/android_perf.h new file mode 100644 index 00000000..f157ad22 --- /dev/null +++ b/native_render/android_perf.h @@ -0,0 +1,140 @@ +#pragma once +// Android frame-rate and CPU-clock plumbing for the live client's render loop. +// +// - display_refresh_rate(): the panel's refresh rate now (MainActivity.currentRefreshRate), and +// render_rate(): the app-vsync rate the system actually gives the app (MainActivity.currentRenderRate), +// for the perf log's refresh_hz / render_hz columns. +// - request_frame_rate(): ANativeWindow_setFrameRate (API 30), the surface's preferred rate. The display +// mode itself is chosen by MainActivity (--refresh-rate -> preferredDisplayModeId); the vendor +// (ColorOS etc.) can still cap an app it does not know. +// - PerformanceHint: ADPF APerformanceHint (API 33). The loop reports each frame's CPU work against the +// frame budget, so the governor raises the clocks before a frame misses instead of idling the CPU at +// its lowest step between short bursts (the 556-748 MHz seen in perf-20260929-153205.csv). +// +// Every NDK entry point is looked up at run time: minSdk is 24. On other platforms all of this is a no-op. + +#include + +#include +#include + +#ifdef __ANDROID__ +#include +#include +#include +#include +#endif + +namespace android_perf { + +#ifdef __ANDROID__ +inline void* libandroid() { + static void* lib = dlopen("libandroid.so", RTLD_NOW); + return lib; +} + +inline double call_activity_float(const char* name) { + auto* env = static_cast(SDL_GetAndroidJNIEnv()); + auto activity = static_cast(SDL_GetAndroidActivity()); + if (!env || !activity) return -1.0; + double hz = -1.0; + jclass cls = env->GetObjectClass(activity); + if (jmethodID method = env->GetMethodID(cls, name, "()F")) + hz = env->CallFloatMethod(activity, method); + if (env->ExceptionCheck()) { + env->ExceptionClear(); + hz = -1.0; + } + env->DeleteLocalRef(cls); + env->DeleteLocalRef(activity); + return hz; +} + +inline double display_refresh_rate() { return call_activity_float("currentRefreshRate"); } +// The app-vsync rate (Choreographer); -1 until the first measurement. +inline double render_rate() { return call_activity_float("currentRenderRate"); } +inline double battery_celsius() { return call_activity_float("batteryTemperature"); } + +// false when the call is unavailable (API < 30) or rejected. +inline bool request_frame_rate(SDL_Window* window, float fps) { + using SetFrameRate = int32_t (*)(ANativeWindow*, float, int8_t); + static const auto set_frame_rate = + libandroid() ? reinterpret_cast(dlsym(libandroid(), "ANativeWindow_setFrameRate")) : nullptr; + if (!set_frame_rate || !window) return false; + auto* native = static_cast(SDL_GetPointerProperty( + SDL_GetWindowProperties(window), SDL_PROP_WINDOW_ANDROID_WINDOW_POINTER, nullptr)); + if (!native) return false; + // ANATIVEWINDOW_FRAME_RATE_COMPATIBILITY_DEFAULT: a game, not fixed-rate video. + return set_frame_rate(native, fps, 0) == 0; +} + +class PerformanceHint { +public: + // `thread_ids`: the threads that do each frame's work (render thread + script thread). + bool open(const std::vector& thread_ids, std::int64_t target_ns) { + void* lib = libandroid(); + if (!lib || thread_ids.empty()) return false; + get_manager_ = reinterpret_cast(dlsym(lib, "APerformanceHint_getManager")); + create_ = reinterpret_cast(dlsym(lib, "APerformanceHint_createSession")); + update_target_ = reinterpret_cast(dlsym(lib, "APerformanceHint_updateTargetWorkDuration")); + report_ = reinterpret_cast(dlsym(lib, "APerformanceHint_reportActualWorkDuration")); + close_ = reinterpret_cast(dlsym(lib, "APerformanceHint_closeSession")); + if (!get_manager_ || !create_ || !report_ || !close_) return false; + void* manager = get_manager_(); + if (!manager) return false; + session_ = create_(manager, thread_ids.data(), thread_ids.size(), target_ns); + target_ns_ = target_ns; + return session_ != nullptr; + } + + ~PerformanceHint() { + if (session_ && close_) close_(session_); + } + + bool active() const { return session_ != nullptr; } + + void set_target(std::int64_t target_ns) { + if (!session_ || !update_target_ || target_ns == target_ns_ || target_ns <= 0) return; + update_target_(session_, target_ns); + target_ns_ = target_ns; + } + + void report(std::int64_t actual_ns) { + if (session_ && actual_ns > 0) report_(session_, actual_ns); + } + +private: + using GetManager = void* (*)(); + using CreateSession = void* (*)(void*, const int32_t*, size_t, int64_t); + using UpdateTarget = int (*)(void*, int64_t); + using Report = int (*)(void*, int64_t); + using Close = void (*)(void*); + GetManager get_manager_ = nullptr; + CreateSession create_ = nullptr; + UpdateTarget update_target_ = nullptr; + Report report_ = nullptr; + Close close_ = nullptr; + void* session_ = nullptr; + std::int64_t target_ns_ = 0; +}; + +inline int32_t current_thread_id() { return int32_t(gettid()); } +#else +inline double display_refresh_rate() { + const SDL_DisplayMode* mode = SDL_GetCurrentDisplayMode(SDL_GetPrimaryDisplay()); + return mode ? double(mode->refresh_rate) : -1.0; +} +inline double render_rate() { return display_refresh_rate(); } +inline double battery_celsius() { return -1.0; } +inline bool request_frame_rate(SDL_Window*, float) { return false; } +class PerformanceHint { +public: + bool open(const std::vector&, std::int64_t) { return false; } + bool active() const { return false; } + void set_target(std::int64_t) {} + void report(std::int64_t) {} +}; +inline int32_t current_thread_id() { return 0; } +#endif + +} // namespace android_perf diff --git a/native_render/draw_capture.h b/native_render/draw_capture.h index d02181a8..8783634c 100644 --- a/native_render/draw_capture.h +++ b/native_render/draw_capture.h @@ -28,7 +28,7 @@ namespace native_draw_capture { struct Capture { - std::uint32_t version = 9; + std::uint32_t version = 10; std::vector draws; std::unordered_map> textures; std::uint32_t ui_width = 960; @@ -114,7 +114,7 @@ inline void write( std::ofstream file(path, std::ios::binary | std::ios::trunc); if (!file) throw std::runtime_error("cannot create native draw capture: " + path); const std::uint32_t magic = 0x4d544452; // MTDR - const std::uint32_t version = 9; + const std::uint32_t version = 10; const auto count = static_cast(draws.size()); write_scalar(file, magic); write_scalar(file, version); write_scalar(file, count); for (const auto& draw : draws) { @@ -193,6 +193,9 @@ inline void write( write_vector(file, draw.bone_indices); write_vector(file, draw.bone_weights); write_vector(file, draw.bone_matrices); + write_scalar(file, draw.render_target); + write_scalar(file, draw.target_width); + write_scalar(file, draw.target_height); } const auto texture_count = static_cast(textures.size()); write_scalar(file, texture_count); @@ -239,7 +242,7 @@ inline Capture read_capture(const std::string& path) { if (!file) throw std::runtime_error("cannot open native draw capture: " + path); std::uint32_t magic = 0, version = 0, count = 0; read_scalar(file, magic); read_scalar(file, version); read_scalar(file, count); - if (magic != 0x4d544452 || (version < 1 || version > 9) || count > 10'000) + if (magic != 0x4d544452 || (version < 1 || version > 10) || count > 10'000) throw std::runtime_error("unsupported native draw capture format"); Capture capture; capture.version = version; @@ -348,6 +351,11 @@ inline Capture read_capture(const std::string& path) { read_vector(file, draw.bone_weights); read_vector(file, draw.bone_matrices); } + if (version >= 10) { + read_scalar(file, draw.render_target); + read_scalar(file, draw.target_width); + read_scalar(file, draw.target_height); + } } if (version >= 3) { std::uint32_t texture_count = 0; diff --git a/native_render/main.cpp b/native_render/main.cpp index 7cbf7e6a..5e9c2d87 100644 --- a/native_render/main.cpp +++ b/native_render/main.cpp @@ -3,6 +3,8 @@ #include "draw_capture.h" #include "dxt.h" #include "touch_controller.h" +#include "perf_log.h" +#include "android_perf.h" #ifdef MT_NATIVE_HAS_LIVE_CLIENT #include "platform/MilesLib/AudioCommands.h" @@ -751,6 +753,7 @@ std::vector read_spirv(const char* name) { } class VulkanWindow { + struct FrameSlot; struct GeometryId { std::uint64_t key = 0, signature = 0; bool operator==(const GeometryId&) const = default; @@ -780,6 +783,19 @@ class VulkanWindow { VkDescriptorSet set = VK_NULL_HANDLE; std::uint64_t last_used_frame = 0; }; + // An offscreen D3D render-target texture (the 40250 character shadow map, "rt::x"): + // drawn by the frame's offscreen pass ahead of the back-buffer pass and sampled by that pass. + // Between frames the colour image stays SHADER_READ_ONLY_OPTIMAL and the depth image + // DEPTH_STENCIL_ATTACHMENT_OPTIMAL; the render pass keeps (LOAD/STORE) both, like a D3D surface. + struct RenderTarget { + GpuTexture texture; // colour image, view and descriptors used for sampling + VkImage depth_image = VK_NULL_HANDLE; + VkDeviceMemory depth_memory = VK_NULL_HANDLE; + VkImageView depth_view = VK_NULL_HANDLE; + VkFramebuffer framebuffer = VK_NULL_HANDLE; + std::uint32_t width = 0, height = 0; + bool initialized = false; // cleared to white / depth 1 by the first frame that uses it + }; struct PreparedDraw { Geometry* geometry = nullptr; VkPipeline pipeline = VK_NULL_HANDLE; @@ -788,6 +804,8 @@ class VulkanWindow { FixedFunctionState fixed_state{}; std::uint32_t state_offset = 0; VkViewport viewport{}; + // Offscreen render target of the draw, or null for the back buffer. + RenderTarget* target = nullptr; // IDirect3DDevice8::Clear recorded in draw order; geometry is null for these. std::uint32_t clear_flags = 0; VkClearValue clear_color{}; @@ -810,16 +828,22 @@ class VulkanWindow { static constexpr std::size_t kMaxUiIndicesPerFrame = 98304; static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex); static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t); + static constexpr VkDeviceSize kStagingRingBytes = 4ull * 1024 * 1024; + static constexpr VkDeviceSize kStagingRingMaxBytes = 64ull * 1024 * 1024; public: struct Timings { double sync_ms = 0; + double fence_ms = 0; // inside sync_ms: waiting for the frame slot's previous GPU work double prepare_ms = 0; double steady_prepare_ms = 0; double submit_ms = 0; double present_ms = 0; double gpu_ms = 0; std::uint64_t gpu_samples = 0; + // Inside prepare_ms: creating geometry buffers and the blocking texture uploads. + double geometry_upload_ms = 0; + double texture_upload_ms = 0; }; #ifdef MT_NATIVE_HAS_LIVE_CLIENT @@ -907,22 +931,25 @@ public: pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; pool_info.queueFamilyIndex = queue_family_; check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool"); - VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; - alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1; - check(vkAllocateCommandBuffers(device_, &alloc, &command_), "vkAllocateCommandBuffers"); VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO}; query_info.queryType = VK_QUERY_TYPE_TIMESTAMP; - query_info.queryCount = 2; + query_info.queryCount = 2 * kMaxFramesInFlight; check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool"); + VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; + for (std::uint32_t i = 0; i < kMaxFramesInFlight; ++i) { + auto& slot = frames_[i]; + VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &alloc, &slot.command), "vkAllocateCommandBuffers"); + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &slot.acquire), "vkCreateSemaphore acquire"); + check(vkCreateFence(device_, &fence_info, nullptr, &slot.fence), "vkCreateFence"); + slot.query_base = 2 * i; + } create_swapchain(); create_descriptors_and_buffers(); create_render_pass_and_layout(); - VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; - check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &acquire_), "vkCreateSemaphore acquire"); - check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &rendered_), "vkCreateSemaphore rendered"); - VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; - fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; - check(vkCreateFence(device_, &fence_info, nullptr, &fence_), "vkCreateFence"); touch_controller.update_screen_size(int(width()), int(height())); } @@ -933,21 +960,29 @@ public: if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_); } if (device_) vkDeviceWaitIdle(device_); - if (bone_mapped_) vkUnmapMemory(device_, bone_memory_); - if (bone_buffer_) vkDestroyBuffer(device_, bone_buffer_, nullptr); - if (bone_memory_) vkFreeMemory(device_, bone_memory_, nullptr); - if (state_mapped_) vkUnmapMemory(device_, state_memory_); - if (state_buffer_) vkDestroyBuffer(device_, state_buffer_, nullptr); - if (state_memory_) vkFreeMemory(device_, state_memory_, nullptr); - if (ui_mapped_) vkUnmapMemory(device_, ui_memory_); - if (ui_buffer_) vkDestroyBuffer(device_, ui_buffer_, nullptr); - if (ui_memory_) vkFreeMemory(device_, ui_memory_, nullptr); + for (auto& slot : frames_) { + release_retired_buffers(slot); + if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); + if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); + if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); + if (slot.bone_mapped) vkUnmapMemory(device_, slot.bone_memory); + if (slot.bone_buffer) vkDestroyBuffer(device_, slot.bone_buffer, nullptr); + if (slot.bone_memory) vkFreeMemory(device_, slot.bone_memory, nullptr); + if (slot.state_mapped) vkUnmapMemory(device_, slot.state_memory); + if (slot.state_buffer) vkDestroyBuffer(device_, slot.state_buffer, nullptr); + if (slot.state_memory) vkFreeMemory(device_, slot.state_memory, nullptr); + if (slot.ui_mapped) vkUnmapMemory(device_, slot.ui_memory); + if (slot.ui_buffer) vkDestroyBuffer(device_, slot.ui_buffer, nullptr); + if (slot.ui_memory) vkFreeMemory(device_, slot.ui_memory, nullptr); + if (slot.fence) vkDestroyFence(device_, slot.fence, nullptr); + if (slot.acquire) vkDestroySemaphore(device_, slot.acquire, nullptr); + } if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr); - if (fence_) vkDestroyFence(device_, fence_, nullptr); - if (acquire_) vkDestroySemaphore(device_, acquire_, nullptr); - if (rendered_) vkDestroySemaphore(device_, rendered_, nullptr); + for (VkSemaphore semaphore : rendered_) vkDestroySemaphore(device_, semaphore, nullptr); for (auto& entry : geometries_) release_geometry(entry.second); for (auto& entry : textures_) release_texture(entry.second); + for (auto& entry : render_targets_) release_render_target(entry.second); + if (offscreen_pass_) vkDestroyRenderPass(device_, offscreen_pass_, nullptr); release_texture(fallback_texture_); for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr); @@ -1035,6 +1070,9 @@ public: void set_safe_inset_px(int px) { safe_inset_px_ = std::max(0, px); } // Test builds: frames per second in the top-left corner (--show-fps). void set_show_fps(bool show) { show_fps_ = show; } + SDL_Window* sdl_window() const { return window_; } + // The last render()'s blocking time: slot fence + swapchain acquire. + double last_sync_ms() const { return last_sync_ms_; } // The same inset in logical UI pixels: the 40250 window layers are shifted and narrowed by it. int ui_safe_inset() const { int pw = 0, ph = 0; @@ -1376,7 +1414,7 @@ public: } void reset_timings() { - collect_pending_gpu_timestamp(); + finish_gpu_timings(); timings_ = {}; timed_frames_ = 0; upload_count_ = 0; @@ -1386,11 +1424,19 @@ public: } void finish_gpu_timings() { - if (!has_pending_query_) return; - check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences finish"); - collect_pending_gpu_timestamp(); + for (auto& slot : frames_) { + if (!slot.has_pending_query) continue; + check(vkWaitForFences(device_, 1, &slot.fence, VK_TRUE, UINT64_MAX), "vkWaitForFences finish"); + collect_pending_gpu_timestamp(slot); + } } + // 1 = the old serial loop (CPU waits for the previous frame's GPU work), 2 = CPU/GPU overlap. + void set_frames_in_flight(std::uint32_t count) { + frames_in_flight_ = std::clamp(count, 1, kMaxFramesInFlight); + } + std::uint32_t frames_in_flight() const { return frames_in_flight_; } + void render( const std::vector& draws, const std::unordered_map>& capture_textures, @@ -1399,11 +1445,14 @@ public: const std::vector& ui_commands = {}) { if (extent_.width == 0 || extent_.height == 0) return; const auto start = std::chrono::steady_clock::now(); - check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences"); - collect_pending_gpu_timestamp(); + f_ = &frames_[frame_slot_ % frames_in_flight_]; + check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences"); + collect_pending_gpu_timestamp(*f_); + reset_staging(*f_); + const auto fenced = std::chrono::steady_clock::now(); prune_textures(); std::uint32_t image_index = 0; - const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, acquire_, VK_NULL_HANDLE, &image_index); + const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, f_->acquire, VK_NULL_HANDLE, &image_index); if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { recreate_swapchain(); return; @@ -1412,9 +1461,26 @@ public: const auto synchronized = std::chrono::steady_clock::now(); ++frame_number_; ++timed_frames_; + while (rendered_.size() < swapchain_images_.size()) { + VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + VkSemaphore semaphore = VK_NULL_HANDLE; + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &semaphore), "vkCreateSemaphore rendered"); + rendered_.push_back(semaphore); + } + // Texture uploads of this frame are recorded ahead of the render pass (upload_texture). + check(vkResetCommandBuffer(f_->command, 0), "vkResetCommandBuffer"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(f_->command, &begin), "vkBeginCommandBuffer"); + vkCmdResetQueryPool(f_->command, query_pool_, f_->query_base, 2); + vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, f_->query_base); + f_->recording = true; std::vector prepared_draws; prepared_draws.reserve(draws.size()); + // Draws into offscreen render targets (the shadow map), in recorded order, drawn before the + // back-buffer pass that samples them. + std::vector offscreen_draws; std::unordered_set active_keys; std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; std::uint32_t bone_cursor = 0; @@ -1429,6 +1495,23 @@ public: ensure_bone_capacity(required_bones); for (const auto& draw : draws) { + RenderTarget* target = nullptr; + if (draw.render_target) { + target = get_render_target( + Render3DRenderTargetName(draw.render_target, draw.target_width, draw.target_height)); + if (!target) continue; + } + if (draw.clear_flags && target) { + PreparedDraw clear{}; + clear.target = target; + clear.clear_flags = draw.clear_flags; + const auto color = unpack_argb(draw.clear_color); + clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; + clear.clear_depth.depthStencil = {draw.clear_z, 0}; + clear.clear_rect = {{0, 0}, {target->width, target->height}}; + offscreen_draws.push_back(clear); + continue; + } if (draw.clear_flags) { PreparedDraw clear{}; clear.clear_flags = draw.clear_flags; @@ -1460,11 +1543,11 @@ public: if (draw.geometry_key) active_keys.insert(draw.geometry_key); float skin_offset_encoded = 0.0f; - if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && bone_mapped_) { + if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && f_->bone_mapped) { const auto bone_count = static_cast(draw.bone_matrices.size() / 16); - if (std::size_t(bone_cursor) + bone_count <= bone_capacity_) { + if (std::size_t(bone_cursor) + bone_count <= f_->bone_capacity) { std::memcpy( - bone_mapped_ + std::size_t(bone_cursor) * 16, + f_->bone_mapped + std::size_t(bone_cursor) * 16, draw.bone_matrices.data(), draw.bone_matrices.size() * sizeof(float)); skin_offset_encoded = float(bone_cursor + 1u); @@ -1490,7 +1573,7 @@ public: } } const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, - static_cast(draw.z_func)); + static_cast(draw.z_func), target != nullptr); // D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples. const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1; const auto descriptor = get_texture_descriptor( @@ -1507,6 +1590,7 @@ public: -1, 1, 0, 1}; } PreparedDraw prepared_draw{}; + prepared_draw.target = target; prepared_draw.geometry = &geometry; prepared_draw.pipeline = pipeline; prepared_draw.descriptor = descriptor; @@ -1516,7 +1600,14 @@ public: prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw); if (draw.pretransformed && draw.ui_offset_x != 0) prepared_draw.viewport.x += draw.ui_offset_x * float(extent_.width) / float(logical_width()); - prepared_draws.push_back(prepared_draw); + if (target) { + // D3DVIEWPORT8 in the target's own pixels. + prepared_draw.viewport = draw.viewport[2] > 0 && draw.viewport[3] > 0 + ? VkViewport{draw.viewport[0], draw.viewport[1], draw.viewport[2], draw.viewport[3], + draw.viewport_z[0], draw.viewport_z[1]} + : VkViewport{0, 0, float(target->width), float(target->height), 0, 1}; + offscreen_draws.push_back(prepared_draw); + } else prepared_draws.push_back(prepared_draw); vertex_count += geometry.vertex_count; index_count += geometry.index_count; } @@ -1531,7 +1622,7 @@ public: std::vector ui_batches; std::size_t ui_quad_count = 0; - if (ui_mapped_) { + if (f_->ui_mapped) { const uint32_t target_ui_w = ui_width ? ui_width : 960; const uint32_t target_ui_h = ui_height ? ui_height : 640; update_fps_counter(); @@ -1555,85 +1646,49 @@ public: } } - ensure_state_capacity(prepared_draws.size() + ui_batches.size()); + ensure_state_capacity(prepared_draws.size() + offscreen_draws.size() + ui_batches.size()); std::size_t state_index = 0; + for (auto& draw : offscreen_draws) { + if (!draw.geometry) continue; + draw.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + ++state_index; + } + last_offscreen_draw_count_ = offscreen_draws.size(); for (auto& draw : prepared_draws) { if (!draw.geometry) continue; draw.state_offset = static_cast(state_index * state_stride_); - std::memcpy(state_mapped_ + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); ++state_index; } const FixedFunctionState ui_state{}; for (auto& batch : ui_batches) { batch.state_offset = static_cast(state_index * state_stride_); - std::memcpy(state_mapped_ + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); + std::memcpy(f_->state_mapped + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); ++state_index; } const auto prepared = std::chrono::steady_clock::now(); - check(vkResetFences(device_, 1, &fence_), "vkResetFences"); - check(vkResetCommandBuffer(command_, 0), "vkResetCommandBuffer"); - VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; - check(vkBeginCommandBuffer(command_, &begin), "vkBeginCommandBuffer"); - vkCmdResetQueryPool(command_, query_pool_, 0, 2); - vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, 0); + f_->recording = false; + check(vkResetFences(device_, 1, &f_->fence), "vkResetFences"); VkClearValue clears[3]{}; // 40250 clears the back buffer to 0xff000000 once at device creation and afterwards // only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer). clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}}; clears[1].depthStencil = {1.0f, 0}; - VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; - pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); - pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears; - vkCmdBeginRenderPass(command_, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); - VkPipeline current_pipeline = VK_NULL_HANDLE; VkDescriptorSet current_descriptor = VK_NULL_HANDLE; VkViewport current_viewport{-1, -1, -1, -1, -1, -1}; auto set_viewport = [&](const VkViewport& viewport) { if (std::memcmp(&viewport, ¤t_viewport, sizeof(viewport)) == 0) return; - vkCmdSetViewport(command_, 0, 1, &viewport); + vkCmdSetViewport(f_->command, 0, 1, &viewport); current_viewport = viewport; }; - auto record_ui_pass = [&](bool behind_3d) { - bool ui_bound = false; - set_viewport(full_viewport()); - for (const auto& batch : ui_batches) { - if (batch.behind_3d != behind_3d) continue; - if (!ui_bound) { - const VkDeviceSize v_offset = 0; - vkCmdBindVertexBuffers(command_, 0, 1, &ui_buffer_, &v_offset); - vkCmdBindIndexBuffer(command_, ui_buffer_, ui_vertex_bytes_, VK_INDEX_TYPE_UINT32); - ui_bound = true; - } - if (batch.pipeline != current_pipeline) { - vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); - current_pipeline = batch.pipeline; - } - if (batch.descriptor != current_descriptor) { - vkCmdBindDescriptorSets( - command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); - current_descriptor = batch.descriptor; - } - vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, - &bone_descriptor_set_, 1, &batch.state_offset); - vkCmdPushConstants( - command_, - layout_, - VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, - 0, - sizeof(batch.constants), - &batch.constants); - vkCmdDrawIndexed(command_, batch.index_count, 1, batch.first_index, 0, 0); - } - }; - - record_ui_pass(true); - - for (const auto& prepared_draw : prepared_draws) { + // One prepared draw or Clear inside the current render pass. + auto record_prepared = [&](const PreparedDraw& prepared_draw) { if (!prepared_draw.geometry) { VkClearAttachment attachments[2]{}; std::uint32_t attachment_count = 0; @@ -1650,38 +1705,96 @@ public: } const VkClearRect rect{prepared_draw.clear_rect, 0, 1}; if (attachment_count && rect.rect.extent.width && rect.rect.extent.height) - vkCmdClearAttachments(command_, attachment_count, attachments, 1, &rect); - continue; + vkCmdClearAttachments(f_->command, attachment_count, attachments, 1, &rect); + return; } set_viewport(prepared_draw.viewport); if (prepared_draw.pipeline != current_pipeline) { - vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); + vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); current_pipeline = prepared_draw.pipeline; } if (prepared_draw.descriptor != current_descriptor) { vkCmdBindDescriptorSets( - command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); + f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); current_descriptor = prepared_draw.descriptor; } - vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, - &bone_descriptor_set_, 1, &prepared_draw.state_offset); + vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &f_->bone_descriptor_set, 1, &prepared_draw.state_offset); const auto& geometry = *prepared_draw.geometry; const VkDeviceSize vertex_offset = 0; - vkCmdBindVertexBuffers(command_, 0, 1, &geometry.buffer, &vertex_offset); - vkCmdBindIndexBuffer(command_, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); + vkCmdBindVertexBuffers(f_->command, 0, 1, &geometry.buffer, &vertex_offset); + vkCmdBindIndexBuffer(f_->command, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); vkCmdPushConstants( - command_, + f_->command, layout_, VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, 0, sizeof(prepared_draw.constants), &prepared_draw.constants); - vkCmdDrawIndexed(command_, geometry.index_count, 1, 0, 0, 0); + vkCmdDrawIndexed(f_->command, geometry.index_count, 1, 0, 0, 0); + }; + + // Render targets first used this frame (drawn or only sampled) start cleared. + for (auto& entry : render_targets_) + if (!entry.second.initialized) initialize_render_target(entry.second); + // Offscreen passes: one per run of draws into the same target. + for (std::size_t i = 0; i < offscreen_draws.size();) { + RenderTarget* target = offscreen_draws[i].target; + VkRenderPassBeginInfo offscreen_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + offscreen_begin.renderPass = offscreen_pass_; + offscreen_begin.framebuffer = target->framebuffer; + offscreen_begin.renderArea.extent = {target->width, target->height}; + vkCmdBeginRenderPass(f_->command, &offscreen_begin, VK_SUBPASS_CONTENTS_INLINE); + for (; i < offscreen_draws.size() && offscreen_draws[i].target == target; ++i) + record_prepared(offscreen_draws[i]); + vkCmdEndRenderPass(f_->command); } + VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); + pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears; + vkCmdBeginRenderPass(f_->command, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); + + auto record_ui_pass = [&](bool behind_3d) { + bool ui_bound = false; + set_viewport(full_viewport()); + for (const auto& batch : ui_batches) { + if (batch.behind_3d != behind_3d) continue; + if (!ui_bound) { + const VkDeviceSize v_offset = 0; + vkCmdBindVertexBuffers(f_->command, 0, 1, &f_->ui_buffer, &v_offset); + vkCmdBindIndexBuffer(f_->command, f_->ui_buffer, f_->ui_vertex_bytes, VK_INDEX_TYPE_UINT32); + ui_bound = true; + } + if (batch.pipeline != current_pipeline) { + vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); + current_pipeline = batch.pipeline; + } + if (batch.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); + current_descriptor = batch.descriptor; + } + vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &f_->bone_descriptor_set, 1, &batch.state_offset); + vkCmdPushConstants( + f_->command, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(batch.constants), + &batch.constants); + vkCmdDrawIndexed(f_->command, batch.index_count, 1, batch.first_index, 0, 0); + } + }; + + record_ui_pass(true); + + for (const auto& prepared_draw : prepared_draws) record_prepared(prepared_draw); + record_ui_pass(false); - vkCmdEndRenderPass(command_); + vkCmdEndRenderPass(f_->command); VkBuffer readback = VK_NULL_HANDLE; VkDeviceMemory readback_memory = VK_NULL_HANDLE; const bool screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_; @@ -1706,41 +1819,101 @@ public: barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; barrier.image = swapchain_images_.at(image_index); barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(command_, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); VkBufferImageCopy region{}; region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; region.imageExtent = {extent_.width, extent_.height, 1}; - vkCmdCopyImageToBuffer(command_, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, ®ion); + vkCmdCopyImageToBuffer(f_->command, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, ®ion); barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; barrier.dstAccessMask = 0; barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - vkCmdPipelineBarrier(command_, VK_PIPELINE_STAGE_TRANSFER_BIT, + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); } - vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, 1); - check(vkEndCommandBuffer(command_), "vkEndCommandBuffer"); - has_pending_query_ = true; + // Debug: MT_DUMP_RT=prefix writes each render target of the screenshot frame to _.pgm. + struct RtDump { VkBuffer buffer; VkDeviceMemory memory; std::uint32_t w, h; }; + std::vector rt_dumps; + const char* dump_rt = std::getenv("MT_DUMP_RT"); + if (screenshot && dump_rt && *dump_rt) { + const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; + for (auto& entry : render_targets_) { + RenderTarget& rt = entry.second; + RtDump dump{VK_NULL_HANDLE, VK_NULL_HANDLE, rt.width, rt.height}; + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = VkDeviceSize(rt.width) * rt.height * texel; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; + check(vkCreateBuffer(device_, &info, nullptr, &dump.buffer), "vkCreateBuffer rt dump"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, dump.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &dump.memory), "vkAllocateMemory rt dump"); + check(vkBindBufferMemory(device_, dump.buffer, dump.memory, 0), "vkBindBufferMemory rt dump"); + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_SHADER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.image = rt.texture.image; + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + VkBufferImageCopy region{}; + region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; + region.imageExtent = {rt.width, rt.height, 1}; + vkCmdCopyImageToBuffer(f_->command, rt.texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dump.buffer, 1, ®ion); + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + rt_dumps.push_back(dump); + } + } + vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, f_->query_base + 1); + check(vkEndCommandBuffer(f_->command), "vkEndCommandBuffer"); + f_->has_pending_query = true; const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; - submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &acquire_; submit.pWaitDstStageMask = &stage; - submit.commandBufferCount = 1; submit.pCommandBuffers = &command_; - submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_; - check(vkQueueSubmit(queue_, 1, &submit, fence_), "vkQueueSubmit"); + submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &f_->acquire; submit.pWaitDstStageMask = &stage; + submit.commandBufferCount = 1; submit.pCommandBuffers = &f_->command; + submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_[image_index]; + check(vkQueueSubmit(queue_, 1, &submit, f_->fence), "vkQueueSubmit"); + ++frame_slot_; if (screenshot) { - check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences readback"); + check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences readback"); void* mapped = nullptr; check(vkMapMemory(device_, readback_memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback"); write_bmp(screenshot_path_, static_cast(mapped)); vkUnmapMemory(device_, readback_memory); + const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; + for (std::size_t n = 0; n < rt_dumps.size(); ++n) { + const auto& dump = rt_dumps[n]; + void* data = nullptr; + check(vkMapMemory(device_, dump.memory, 0, VK_WHOLE_SIZE, 0, &data), "vkMapMemory rt dump"); + const auto* bytes = static_cast(data); + std::ofstream out(std::string(dump_rt) + "_" + std::to_string(n) + ".pgm", std::ios::binary); + out << "P5\n" << dump.w << " " << dump.h << "\n255\n"; + for (std::size_t i = 0; i < std::size_t(dump.w) * dump.h; ++i) { + std::uint8_t g = texel == 2 ? std::uint8_t(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31) : bytes[i * 4]; + out.put(char(g)); + } + vkUnmapMemory(device_, dump.memory); + vkDestroyBuffer(device_, dump.buffer, nullptr); + vkFreeMemory(device_, dump.memory, nullptr); + } vkDestroyBuffer(device_, readback, nullptr); vkFreeMemory(device_, readback_memory, nullptr); } const auto submitted = std::chrono::steady_clock::now(); VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; - present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_; + present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_[image_index]; present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; const auto presented = vkQueuePresentKHR(queue_, &present); // An Android surface can continuously report SUBOPTIMAL while the app deliberately @@ -1753,7 +1926,9 @@ public: } const auto presented_at = std::chrono::steady_clock::now(); const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); - timings_.sync_ms += std::chrono::duration(synchronized - start).count(); + last_sync_ms_ = std::chrono::duration(synchronized - start).count(); + timings_.sync_ms += last_sync_ms_; + timings_.fence_ms += std::chrono::duration(fenced - start).count(); timings_.prepare_ms += prepare_ms; if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); @@ -1785,6 +1960,28 @@ public: } } Timings timings() const { return timings_; } + native_perf::RendererTotals perf_totals() const { + native_perf::RendererTotals t; + t.sync_ms = timings_.sync_ms; + t.fence_ms = timings_.fence_ms; + t.prepare_ms = timings_.prepare_ms; + t.geometry_upload_ms = timings_.geometry_upload_ms; + t.texture_upload_ms = timings_.texture_upload_ms; + t.submit_ms = timings_.submit_ms; + t.present_ms = timings_.present_ms; + t.gpu_ms = timings_.gpu_ms; + t.gpu_samples = timings_.gpu_samples; + t.uploads = upload_count_; + t.uploaded_bytes = uploaded_bytes_; + t.texture_uploads = texture_upload_count_; + t.texture_bytes = texture_uploaded_bytes_; + t.draws = last_draw_count_; + t.skinned_draws = last_skinned_draw_count_; + t.ui_batches = last_ui_batch_count_; + t.vertices = last_vertex_count_; + t.last_texture = &last_texture_upload_name_; + return t; + } private: void recreate_swapchain() { @@ -1808,12 +2005,13 @@ private: get_pipeline(0, 2, 0); } - void collect_pending_gpu_timestamp() { - if (!has_pending_query_) return; - has_pending_query_ = false; + void collect_pending_gpu_timestamp(FrameSlot& slot) { + if (!slot.has_pending_query) return; + slot.has_pending_query = false; std::uint64_t timestamps[2] = {}; const VkResult res = vkGetQueryPoolResults( - device_, query_pool_, 0, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), VK_QUERY_RESULT_64_BIT); + device_, query_pool_, slot.query_base, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), + VK_QUERY_RESULT_64_BIT); if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) { const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_); timings_.gpu_ms += ns * 1e-6; @@ -1829,15 +2027,15 @@ private: std::vector& batches, std::size_t& out_quad_count) { ensure_ui_capacity(commands.size()); - auto* vertices = reinterpret_cast(ui_mapped_); - auto* indices = reinterpret_cast(ui_mapped_ + ui_vertex_bytes_); + auto* vertices = reinterpret_cast(f_->ui_mapped); + auto* indices = reinterpret_cast(f_->ui_mapped + f_->ui_vertex_bytes); std::uint32_t v_count = 0; std::uint32_t i_count = 0; const float inv_w = 2.0f / float(ui_width); const float inv_h = 2.0f / float(ui_height); for (const auto& cmd : commands) { - if (v_count + 4 > ui_vertex_capacity_ || i_count + 6 > ui_index_capacity_) + if (v_count + 4 > f_->ui_vertex_capacity || i_count + 6 > f_->ui_index_capacity) throw std::runtime_error("UI command buffer capacity exceeded"); if (cmd.kind == UIRenderCommand::Text) continue; @@ -2306,8 +2504,8 @@ private: VkDescriptorPoolSize pool_sizes[3] = { {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, - {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4}, - {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4}}; + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4 * kMaxFramesInFlight}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4 * kMaxFramesInFlight}}; VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; pool_info.maxSets = 8192; @@ -2320,79 +2518,109 @@ private: fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); - void* bone_raw = nullptr; - create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, bone_buffer_, bone_memory_, &bone_raw); - bone_mapped_ = static_cast(bone_raw); - bone_capacity_ = kMaxBonesPerFrame; - for (std::size_t i = 0; i < 16; ++i) bone_mapped_[i] = kIdentityMatrix[i]; - VkPhysicalDeviceProperties properties{}; vkGetPhysicalDeviceProperties(physical_, &properties); const auto alignment = std::max( 1, properties.limits.minStorageBufferOffsetAlignment); state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; - state_capacity_ = 256; - void* state_raw = nullptr; - create_host_buffer(state_capacity_ * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, - state_buffer_, state_memory_, &state_raw); - state_mapped_ = static_cast(state_raw); + staging_alignment_ = std::max(16, properties.limits.optimalBufferCopyOffsetAlignment); - VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - bone_alloc.descriptorPool = descriptor_pool_; - bone_alloc.descriptorSetCount = 1; - bone_alloc.pSetLayouts = &bone_descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &bone_alloc, &bone_descriptor_set_), "vkAllocateDescriptorSets bone"); + for (auto& slot : frames_) { + void* bone_raw = nullptr; + create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, slot.bone_buffer, slot.bone_memory, &bone_raw); + slot.bone_mapped = static_cast(bone_raw); + slot.bone_capacity = kMaxBonesPerFrame; + for (std::size_t i = 0; i < 16; ++i) slot.bone_mapped[i] = kIdentityMatrix[i]; - VkDescriptorBufferInfo buffer_infos[2] = { - {bone_buffer_, 0, kBoneBufferBytes}, - {state_buffer_, 0, sizeof(FixedFunctionState)}}; - VkWriteDescriptorSet writes[2]{}; - for (int binding = 0; binding < 2; ++binding) { - writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[binding].dstSet = bone_descriptor_set_; - writes[binding].dstBinding = static_cast(binding); - writes[binding].descriptorCount = 1; - writes[binding].descriptorType = binding == 0 - ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - writes[binding].pBufferInfo = &buffer_infos[binding]; + slot.state_capacity = 256; + void* state_raw = nullptr; + create_host_buffer(slot.state_capacity * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + slot.state_buffer, slot.state_memory, &state_raw); + slot.state_mapped = static_cast(state_raw); + + VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + bone_alloc.descriptorPool = descriptor_pool_; + bone_alloc.descriptorSetCount = 1; + bone_alloc.pSetLayouts = &bone_descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &bone_alloc, &slot.bone_descriptor_set), "vkAllocateDescriptorSets bone"); + + VkDescriptorBufferInfo buffer_infos[2] = { + {slot.bone_buffer, 0, kBoneBufferBytes}, + {slot.state_buffer, 0, sizeof(FixedFunctionState)}}; + VkWriteDescriptorSet writes[2]{}; + for (int binding = 0; binding < 2; ++binding) { + writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[binding].dstSet = slot.bone_descriptor_set; + writes[binding].dstBinding = static_cast(binding); + writes[binding].descriptorCount = 1; + writes[binding].descriptorType = binding == 0 + ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + writes[binding].pBufferInfo = &buffer_infos[binding]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + + void* ui_raw = nullptr; + create_host_buffer( + kUiBufferBytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, + slot.ui_buffer, + slot.ui_memory, + &ui_raw); + slot.ui_mapped = static_cast(ui_raw); + slot.ui_vertex_capacity = kMaxUiVerticesPerFrame; + slot.ui_index_capacity = kMaxUiIndicesPerFrame; + slot.ui_vertex_bytes = kUiVertexBytes; + slot.staging_wanted = kStagingRingBytes; } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + } - void* ui_raw = nullptr; - create_host_buffer( - kUiBufferBytes, - VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, - ui_buffer_, - ui_memory_, - &ui_raw); - ui_mapped_ = static_cast(ui_raw); - ui_vertex_capacity_ = kMaxUiVerticesPerFrame; - ui_index_capacity_ = kMaxUiIndicesPerFrame; - ui_vertex_bytes_ = kUiVertexBytes; + // The slot's fence has signalled: its oversize staging buffers are free, and the ring restarts + // (grown to the largest frame's uploads seen so far, so texture streaming stays in the ring). + void release_retired_buffers(FrameSlot& slot) { + for (auto [buffer, memory] : slot.retired_buffers) { + vkDestroyBuffer(device_, buffer, nullptr); + vkFreeMemory(device_, memory, nullptr); + } + slot.retired_buffers.clear(); + } + + void reset_staging(FrameSlot& slot) { + release_retired_buffers(slot); + slot.staging_used = 0; + if (slot.staging_wanted <= slot.staging_capacity) return; + if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); + if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); + if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); + void* raw = nullptr; + slot.staging_capacity = std::min(slot.staging_wanted, kStagingRingMaxBytes); + create_host_buffer(slot.staging_capacity, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, + slot.staging_buffer, slot.staging_memory, &raw); + slot.staging_mapped = static_cast(raw); + slot.staging_wanted = slot.staging_capacity; } void ensure_bone_capacity(std::size_t required) { - if (required <= bone_capacity_) return; + if (required <= f_->bone_capacity) return; VkPhysicalDeviceProperties properties{}; vkGetPhysicalDeviceProperties(physical_, &properties); const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); if (required > max_bones || required > UINT32_MAX) throw std::runtime_error("bone palette exceeds device storage-buffer limit"); - std::size_t next = std::min(max_bones, std::max(required, bone_capacity_ * 2)); + std::size_t next = std::min(max_bones, std::max(required, f_->bone_capacity * 2)); VkBuffer buffer = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; void* mapped = nullptr; create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, bone_memory_); - vkDestroyBuffer(device_, bone_buffer_, nullptr); - vkFreeMemory(device_, bone_memory_, nullptr); - bone_buffer_ = buffer; - bone_memory_ = memory; - bone_mapped_ = static_cast(mapped); - bone_capacity_ = next; - VkDescriptorBufferInfo info{bone_buffer_, 0, next * 16 * sizeof(float)}; + vkUnmapMemory(device_, f_->bone_memory); + vkDestroyBuffer(device_, f_->bone_buffer, nullptr); + vkFreeMemory(device_, f_->bone_memory, nullptr); + f_->bone_buffer = buffer; + f_->bone_memory = memory; + f_->bone_mapped = static_cast(mapped); + f_->bone_capacity = next; + VkDescriptorBufferInfo info{f_->bone_buffer, 0, next * 16 * sizeof(float)}; VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; - write.dstSet = bone_descriptor_set_; + write.dstSet = f_->bone_descriptor_set; write.dstBinding = 0; write.descriptorCount = 1; write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; @@ -2401,8 +2629,8 @@ private: } void ensure_state_capacity(std::size_t required) { - if (required <= state_capacity_) return; - const auto next = std::max(required, state_capacity_ * 2); + if (required <= f_->state_capacity) return; + const auto next = std::max(required, f_->state_capacity * 2); if (next > UINT32_MAX / state_stride_) throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); VkBuffer buffer = VK_NULL_HANDLE; @@ -2410,16 +2638,16 @@ private: void* mapped = nullptr; create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, state_memory_); - vkDestroyBuffer(device_, state_buffer_, nullptr); - vkFreeMemory(device_, state_memory_, nullptr); - state_buffer_ = buffer; - state_memory_ = memory; - state_mapped_ = static_cast(mapped); - state_capacity_ = next; - VkDescriptorBufferInfo info{state_buffer_, 0, sizeof(FixedFunctionState)}; + vkUnmapMemory(device_, f_->state_memory); + vkDestroyBuffer(device_, f_->state_buffer, nullptr); + vkFreeMemory(device_, f_->state_memory, nullptr); + f_->state_buffer = buffer; + f_->state_memory = memory; + f_->state_mapped = static_cast(mapped); + f_->state_capacity = next; + VkDescriptorBufferInfo info{f_->state_buffer, 0, sizeof(FixedFunctionState)}; VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; - write.dstSet = bone_descriptor_set_; + write.dstSet = f_->bone_descriptor_set; write.dstBinding = 1; write.descriptorCount = 1; write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; @@ -2428,9 +2656,9 @@ private: } void ensure_ui_capacity(std::size_t command_count) { - if (command_count <= ui_vertex_capacity_ / 4 && command_count <= ui_index_capacity_ / 6) return; + if (command_count <= f_->ui_vertex_capacity / 4 && command_count <= f_->ui_index_capacity / 6) return; if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); - const std::size_t quads = std::max(command_count, ui_vertex_capacity_ / 2); + const std::size_t quads = std::max(command_count, f_->ui_vertex_capacity / 2); const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) @@ -2440,15 +2668,15 @@ private: void* mapped = nullptr; create_host_buffer(vertex_bytes + index_bytes, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, ui_memory_); - vkDestroyBuffer(device_, ui_buffer_, nullptr); - vkFreeMemory(device_, ui_memory_, nullptr); - ui_buffer_ = buffer; - ui_memory_ = memory; - ui_mapped_ = static_cast(mapped); - ui_vertex_capacity_ = quads * 4; - ui_index_capacity_ = quads * 6; - ui_vertex_bytes_ = vertex_bytes; + vkUnmapMemory(device_, f_->ui_memory); + vkDestroyBuffer(device_, f_->ui_buffer, nullptr); + vkFreeMemory(device_, f_->ui_memory, nullptr); + f_->ui_buffer = buffer; + f_->ui_memory = memory; + f_->ui_mapped = static_cast(mapped); + f_->ui_vertex_capacity = quads * 4; + f_->ui_index_capacity = quads * 6; + f_->ui_vertex_bytes = vertex_bytes; } void create_render_pass_and_layout() { @@ -2512,13 +2740,168 @@ private: check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); get_pipeline(0, 2, 0); + create_offscreen_pass(); } + // The render pass of offscreen render-target textures: colour (R5G6B5 like the D3D surface when + // the device can render to it) + D32 depth, both loaded and stored, colour left shader-readable. + void create_offscreen_pass() { + VkFormatProperties properties{}; + vkGetPhysicalDeviceFormatProperties(physical_, VK_FORMAT_R5G6B5_UNORM_PACK16, &properties); + const VkFormatFeatureFlags wanted = + VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT; + offscreen_format_ = (properties.optimalTilingFeatures & wanted) == wanted + ? VK_FORMAT_R5G6B5_UNORM_PACK16 : VK_FORMAT_R8G8B8A8_UNORM; + VkAttachmentDescription attachments[2]{}; + attachments[0].format = offscreen_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[0].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; + attachments[0].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[0].initialLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + attachments[0].finalLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[1].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; + attachments[1].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[1].initialLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; + VkSubpassDescription subpass{}; + subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; + subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; + subpass.pDepthStencilAttachment = &depth_ref; + VkSubpassDependency dependencies[2]{}; + // Earlier work on the queue -- the previous frame sampling the texture, the previous + // offscreen pass, the first-use clear -- completes before this pass writes it. + dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL; dependencies[0].dstSubpass = 0; + dependencies[0].srcStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | + VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | + VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT; + dependencies[0].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | + VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT; + dependencies[0].dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | + VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT; + dependencies[0].dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | + VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; + // The back-buffer pass samples the result. + dependencies[1].srcSubpass = 0; dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL; + dependencies[1].srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; + dependencies[1].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; + dependencies[1].dstStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT; + dependencies[1].dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; + pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; + pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; + pass_info.dependencyCount = 2; pass_info.pDependencies = dependencies; + check(vkCreateRenderPass(device_, &pass_info, nullptr, &offscreen_pass_), "vkCreateRenderPass offscreen"); + } + + // "rt::x" -> its render target, created on first use (by a draw into it or by a draw + // sampling it); null for a malformed name. + RenderTarget* get_render_target(const std::string& name) { + if (auto it = render_targets_.find(name); it != render_targets_.end()) return &it->second; + unsigned id = 0, w = 0, h = 0; + if (std::sscanf(name.c_str(), "rt:%u:%ux%u", &id, &w, &h) != 3 || !w || !h || w > 4096 || h > 4096) + return nullptr; + RenderTarget& target = render_targets_[name]; + target.width = w; + target.height = h; + auto make_image = [&](VkFormat format, VkImageUsageFlags usage, VkImageAspectFlags aspect, + VkImage& image, VkDeviceMemory& memory, VkImageView& view) { + VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + info.imageType = VK_IMAGE_TYPE_2D; + info.format = format; + info.extent = {w, h, 1}; + info.mipLevels = 1; info.arrayLayers = 1; + info.samples = VK_SAMPLE_COUNT_1_BIT; + info.tiling = VK_IMAGE_TILING_OPTIMAL; + info.usage = usage; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &info, nullptr, &image), "vkCreateImage render target"); + VkMemoryRequirements reqs{}; + vkGetImageMemoryRequirements(device_, image, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory render target"); + check(vkBindImageMemory(device_, image, memory, 0), "vkBindImageMemory render target"); + VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view_info.image = image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = format; + view_info.subresourceRange = {aspect, 0, 1, 0, 1}; + check(vkCreateImageView(device_, &view_info, nullptr, &view), "vkCreateImageView render target"); + }; + make_image(offscreen_format_, + VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT, + VK_IMAGE_ASPECT_COLOR_BIT, target.texture.image, target.texture.memory, target.texture.view); + make_image(VK_FORMAT_D32_SFLOAT, + VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, + VK_IMAGE_ASPECT_DEPTH_BIT, target.depth_image, target.depth_memory, target.depth_view); + const VkImageView views[2] = {target.texture.view, target.depth_view}; + VkFramebufferCreateInfo fb{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + fb.renderPass = offscreen_pass_; + fb.attachmentCount = 2; fb.pAttachments = views; + fb.width = w; fb.height = h; fb.layers = 1; + check(vkCreateFramebuffer(device_, &fb, nullptr, &target.framebuffer), "vkCreateFramebuffer render target"); + target.texture.descriptor = allocate_texture_descriptor_set(target.texture.view, fallback_texture_.view); + target.texture.ui_descriptor = allocate_ui_texture_descriptor_set(target.texture.view, fallback_texture_.view); + return ⌖ + } + + void release_render_target(RenderTarget& target) { + if (target.framebuffer) vkDestroyFramebuffer(device_, target.framebuffer, nullptr); + if (target.depth_view) vkDestroyImageView(device_, target.depth_view, nullptr); + if (target.depth_image) vkDestroyImage(device_, target.depth_image, nullptr); + if (target.depth_memory) vkFreeMemory(device_, target.depth_memory, nullptr); + release_texture(target.texture); + target = {}; + } + + // A new render target starts white with depth 1 (what the CPU surface's first Clear leaves), in + // the layouts offscreen_pass_ expects. Recorded outside any render pass. + void initialize_render_target(RenderTarget& target) { + VkImageMemoryBarrier barriers[2]{}; + for (auto& b : barriers) { + b.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER; + b.srcQueueFamilyIndex = b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + b.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + b.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + b.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + } + barriers[0].image = target.texture.image; + barriers[0].subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + barriers[1].image = target.depth_image; + barriers[1].subresourceRange = {VK_IMAGE_ASPECT_DEPTH_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 2, barriers); + const VkClearColorValue white{{1.0f, 1.0f, 1.0f, 1.0f}}; + const VkClearDepthStencilValue far{1.0f, 0}; + vkCmdClearColorImage(f_->command, target.texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &white, 1, + &barriers[0].subresourceRange); + vkCmdClearDepthStencilImage(f_->command, target.depth_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &far, 1, + &barriers[1].subresourceRange); + for (auto& b : barriers) { + b.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + b.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + } + barriers[0].newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barriers[0].dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_READ_BIT; + barriers[1].newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + barriers[1].dstAccessMask = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | + VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, + 0, 0, nullptr, 0, nullptr, 2, barriers); + target.initialized = true; + } + + // offscreen: for offscreen_pass_ (single-sampled render-target texture) instead of pass_. VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, - bool lines = false, std::uint8_t z_func = 4) { + bool lines = false, std::uint8_t z_func = 4, bool offscreen = false) { const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | - (std::uint32_t(z_func & 0xfu) << 25); + (std::uint32_t(z_func & 0xfu) << 25) | (offscreen ? (1u << 29) : 0u); if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; VkPipelineShaderStageCreateInfo stages[2]{}; @@ -2545,7 +2928,8 @@ private: assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; - VkRect2D scissor{{0, 0}, extent_}; + // An offscreen target is at most the device's 4096 (GetDeviceCaps); the viewport clips inside it. + VkRect2D scissor{{0, 0}, offscreen ? VkExtent2D{4096, 4096} : extent_}; VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; @@ -2561,7 +2945,7 @@ private: raster.lineWidth = 1; VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; - multisample.rasterizationSamples = samples_; + multisample.rasterizationSamples = offscreen ? VK_SAMPLE_COUNT_1_BIT : samples_; VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; @@ -2634,7 +3018,7 @@ private: pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; pipeline_info.pColorBlendState = &blending; pipeline_info.pDynamicState = &dynamic_info; - pipeline_info.layout = layout_; pipeline_info.renderPass = pass_; + pipeline_info.layout = layout_; pipeline_info.renderPass = offscreen ? offscreen_pass_ : pass_; VkPipeline handle = VK_NULL_HANDLE; check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); pipelines_.emplace(key, handle); @@ -2705,6 +3089,10 @@ private: const std::string& texture_name, const std::unordered_map>& capture_textures) { if (texture_name.empty()) return fallback_texture_; + if (texture_name.rfind("rt:", 0) == 0) { + RenderTarget* target = get_render_target(texture_name); + return target ? target->texture : fallback_texture_; + } auto [it, inserted] = textures_.try_emplace(texture_name); it->second.last_used_frame = frame_number_; if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) @@ -2742,6 +3130,7 @@ private: // other files -> D3DXCreateTextureFromFileInMemoryEx(D3DX_DEFAULT) = full chain; // memory images (mem:, CreateTexture(w, h, 1, ...)) -> one level. const bool is_memory_tex = texture_name.rfind("mem:", 0) == 0; + last_texture_upload_name_ = texture_name; upload_texture( decoded.w, decoded.h, decoded.rgba.data(), it->second, true, !is_memory_tex && decoded.file_mip_levels == 0, &decoded.mips); @@ -2839,6 +3228,7 @@ private: bool count_upload, bool generate_mips = true, const std::vector>* file_mips = nullptr) { + const ScopedMs timer(timings_.texture_upload_ms); const bool has_file_mips = !generate_mips && file_mips && !file_mips->empty(); VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; if (has_file_mips) @@ -2846,29 +3236,33 @@ private: const std::uint32_t mip_levels = generate_mips ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) : has_file_mips ? 1u + static_cast(file_mips->size()) : 1u; + // Inside a frame the copy is recorded ahead of that frame's render pass and reads the slot's + // staging ring; the GPU runs it with the frame, no queue wait. Outside a frame (the fallback + // texture at startup) it is a one-off submit that waits. + const bool in_frame = f_->recording; VkBuffer staging_buffer = VK_NULL_HANDLE; VkDeviceMemory staging_memory = VK_NULL_HANDLE; - VkBufferCreateInfo buffer_info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - buffer_info.size = byte_size; - buffer_info.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT; - buffer_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateBuffer(device_, &buffer_info, nullptr, &staging_buffer), "vkCreateBuffer staging"); - VkMemoryRequirements buffer_reqs{}; - vkGetBufferMemoryRequirements(device_, staging_buffer, &buffer_reqs); - VkMemoryAllocateInfo buffer_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - buffer_alloc.allocationSize = buffer_reqs.size; - buffer_alloc.memoryTypeIndex = find_memory_type( - buffer_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &buffer_alloc, nullptr, &staging_memory), "vkAllocateMemory staging"); - check(vkBindBufferMemory(device_, staging_buffer, staging_memory, 0), "vkBindBufferMemory staging"); - void* mapped = nullptr; - check(vkMapMemory(device_, staging_memory, 0, byte_size, 0, &mapped), "vkMapMemory staging"); + VkDeviceSize staging_offset = 0; + std::uint8_t* mapped = nullptr; + const VkDeviceSize aligned_offset = + (f_->staging_used + staging_alignment_ - 1) / staging_alignment_ * staging_alignment_; + if (in_frame && f_->staging_mapped && aligned_offset + byte_size <= f_->staging_capacity) { + staging_buffer = f_->staging_buffer; + staging_offset = aligned_offset; + mapped = f_->staging_mapped + staging_offset; + f_->staging_used = aligned_offset + byte_size; + } else { + if (in_frame) f_->staging_wanted = std::max(f_->staging_wanted, aligned_offset + byte_size); + void* raw = nullptr; + create_host_buffer(byte_size, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, staging_buffer, staging_memory, &raw); + mapped = static_cast(raw); + } std::memcpy(mapped, rgba, VkDeviceSize(width) * height * 4); if (has_file_mips) { - auto* dst = static_cast(mapped) + VkDeviceSize(width) * height * 4; + auto* dst = mapped + VkDeviceSize(width) * height * 4; for (const auto& level : *file_mips) { std::memcpy(dst, level.data(), level.size()); dst += level.size(); } } - vkUnmapMemory(device_, staging_memory); + if (staging_memory) vkUnmapMemory(device_, staging_memory); VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; image_info.imageType = VK_IMAGE_TYPE_2D; @@ -2890,13 +3284,15 @@ private: check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); - VkCommandBuffer upload_cmd = VK_NULL_HANDLE; - VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; - cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; - check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); - VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; - begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; - check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); + VkCommandBuffer upload_cmd = f_->command; + if (!in_frame) { + VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); + } VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; @@ -2914,7 +3310,7 @@ private: VkDeviceSize region_offset = 0; for (std::uint32_t i = 0; i < regions.size(); ++i) { const std::uint32_t lw = std::max(1u, width >> i), lh = std::max(1u, height >> i); - regions[i].bufferOffset = region_offset; + regions[i].bufferOffset = staging_offset + region_offset; regions[i].imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; regions[i].imageExtent = {lw, lh, 1}; region_offset += VkDeviceSize(lw) * lh * 4; @@ -2981,14 +3377,18 @@ private: upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, 0, 0, nullptr, 0, nullptr, 1, &to_shader); - check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); - VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; - submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; - check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); - check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); - vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); - vkDestroyBuffer(device_, staging_buffer, nullptr); - vkFreeMemory(device_, staging_memory, nullptr); + if (in_frame) { + if (staging_memory) f_->retired_buffers.emplace_back(staging_buffer, staging_memory); + } else { + check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; + check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); + check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); + vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); + vkDestroyBuffer(device_, staging_buffer, nullptr); + vkFreeMemory(device_, staging_memory, nullptr); + } VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; @@ -3040,6 +3440,7 @@ private: } void upload_geometry(const Render3DDraw& draw, Geometry& geometry) { + const ScopedMs timer(timings_.geometry_upload_ms); if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) throw std::runtime_error("invalid native geometry array size"); const auto count = draw.positions.size() / 3; @@ -3135,6 +3536,14 @@ private: int fps_frames_ = 0; std::chrono::steady_clock::time_point fps_window_start_{}; + // Adds the scope's wall time to a Timings field. + struct ScopedMs { + explicit ScopedMs(double& into) : into_(into), start_(std::chrono::steady_clock::now()) {} + ~ScopedMs() { into_ += std::chrono::duration(std::chrono::steady_clock::now() - start_).count(); } + double& into_; + std::chrono::steady_clock::time_point start_; + }; + // Presented frames over the last half second. void update_fps_counter() { const auto now = std::chrono::steady_clock::now(); @@ -3180,24 +3589,54 @@ private: VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE; VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE; VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE; - VkDescriptorSet bone_descriptor_set_ = VK_NULL_HANDLE; - VkBuffer bone_buffer_ = VK_NULL_HANDLE; - VkDeviceMemory bone_memory_ = VK_NULL_HANDLE; - float* bone_mapped_ = nullptr; - std::size_t bone_capacity_ = 0; - VkBuffer state_buffer_ = VK_NULL_HANDLE; - VkDeviceMemory state_memory_ = VK_NULL_HANDLE; - std::uint8_t* state_mapped_ = nullptr; + // Everything one frame writes while the GPU may still read the previous frame's copy: with + // kFramesInFlight slots the CPU prepares frame N while the GPU draws frame N-1. A slot is reused + // only after its fence (the submit of frame N - frames_in_flight_) has signalled. + struct FrameSlot { + VkCommandBuffer command = VK_NULL_HANDLE; + VkFence fence = VK_NULL_HANDLE; + VkSemaphore acquire = VK_NULL_HANDLE; + std::uint32_t query_base = 0; + bool has_pending_query = false; + bool recording = false; + VkDescriptorSet bone_descriptor_set = VK_NULL_HANDLE; + VkBuffer bone_buffer = VK_NULL_HANDLE; + VkDeviceMemory bone_memory = VK_NULL_HANDLE; + float* bone_mapped = nullptr; + std::size_t bone_capacity = 0; + VkBuffer state_buffer = VK_NULL_HANDLE; + VkDeviceMemory state_memory = VK_NULL_HANDLE; + std::uint8_t* state_mapped = nullptr; + std::size_t state_capacity = 0; + VkBuffer ui_buffer = VK_NULL_HANDLE; + VkDeviceMemory ui_memory = VK_NULL_HANDLE; + std::uint8_t* ui_mapped = nullptr; + std::size_t ui_vertex_capacity = 0, ui_index_capacity = 0; + VkDeviceSize ui_vertex_bytes = 0; + // Texture uploads recorded into this frame's command buffer copy from here. + VkBuffer staging_buffer = VK_NULL_HANDLE; + VkDeviceMemory staging_memory = VK_NULL_HANDLE; + std::uint8_t* staging_mapped = nullptr; + VkDeviceSize staging_capacity = 0, staging_used = 0, staging_wanted = 0; + // Uploads larger than the ring get their own staging buffer, freed when the slot comes back. + std::vector> retired_buffers; + }; + static constexpr std::uint32_t kMaxFramesInFlight = 2; + std::array frames_{}; + FrameSlot* f_ = &frames_[0]; + std::uint32_t frames_in_flight_ = kMaxFramesInFlight; + std::uint32_t frame_slot_ = 0; + // Signalled by a frame's submit, waited by the present of that swapchain image. + std::vector rendered_; VkDeviceSize state_stride_ = 0; - std::size_t state_capacity_ = 0; - VkBuffer ui_buffer_ = VK_NULL_HANDLE; - VkDeviceMemory ui_memory_ = VK_NULL_HANDLE; - std::uint8_t* ui_mapped_ = nullptr; - std::size_t ui_vertex_capacity_ = 0, ui_index_capacity_ = 0; - VkDeviceSize ui_vertex_bytes_ = 0; + VkDeviceSize staging_alignment_ = 16; GpuTexture fallback_texture_{}; std::unordered_map textures_; std::unordered_map paired_descriptors_; + std::unordered_map render_targets_; + VkRenderPass offscreen_pass_ = VK_NULL_HANDLE; + VkFormat offscreen_format_ = VK_FORMAT_R5G6B5_UNORM_PACK16; + std::size_t last_offscreen_draw_count_ = 0; VkShaderModule vertex_module_ = VK_NULL_HANDLE; VkShaderModule fragment_module_ = VK_NULL_HANDLE; VkPipelineLayout layout_ = VK_NULL_HANDLE; @@ -3207,22 +3646,20 @@ private: std::uint64_t timed_frames_ = 0; std::size_t upload_count_ = 0, uploaded_bytes_ = 0; std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0; + std::string last_texture_upload_name_; VkCommandPool pool_ = VK_NULL_HANDLE; - VkCommandBuffer command_ = VK_NULL_HANDLE; VkQueryPool query_pool_ = VK_NULL_HANDLE; - bool has_pending_query_ = false; bool os_cursor_hidden_ = false; bool hardware_cursor_enabled_ = false; std::array game_cursors_{}; SDL_Cursor* fallback_cursor_ = nullptr; int current_cursor_shape_ = -1; int last_mx_ = 0, last_my_ = 0; - VkSemaphore acquire_ = VK_NULL_HANDLE, rendered_ = VK_NULL_HANDLE; - VkFence fence_ = VK_NULL_HANDLE; std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0; std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0; std::size_t last_vertex_count_ = 0, last_index_count_ = 0; Timings timings_{}; + double last_sync_ms_ = 0; }; std::vector make_test_draws(std::size_t count, std::size_t triangles_per_draw) { @@ -3295,6 +3732,13 @@ void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, // --py-exec: a test hook run once the auto-login run reaches the GameWindow (or, with --login-screen, // once the LoginWindow is up). std::string g_py_exec_after_login; +// --perf-log / MT_PERF_LOG=1: native_render/perf_log.h. +bool g_perf_log = false; +// --fps-cap N: the loop starts a frame at most every 1/N s (0 = as fast as vsync allows). +int g_fps_cap = 0; +// --refresh-rate N: the surface's preferred frame rate (Android ANativeWindow_setFrameRate; MainActivity +// also picks the matching display mode). 0 leaves the system default. +int g_refresh_rate = 0; bool py_exec(const std::string& code) { std::string err; @@ -3384,6 +3828,10 @@ int run_live_client( SetNativeTerrainRenderEnabled(native_terrain); SetGpuSkinningEnabled(gpu_skinning); + // The shadow map is drawn by the GPU (offscreen pass); MT_CPU_SHADOW=1 / --cpu-shadow keeps the + // CPU rasterizer (RecordingDevice::rasterize_shadow) for comparison. + const char* cpu_shadow = std::getenv("MT_CPU_SHADOW"); + SetGpuRenderTargetsEnabled(!(cpu_shadow && *cpu_shadow == '1')); NativeAudioEngine audio; #ifdef __ANDROID__ SDL_Log("live stage: audio initialized"); @@ -3620,15 +4068,65 @@ int run_live_client( } renderer.reset_timings(); - const auto start = std::chrono::steady_clock::now(); + native_perf::PerfLog perf_log; + if (g_perf_log) { + perf_log.open("device=" + renderer.device_name() + " present_mode=" + renderer.present_mode_name() + + " msaa=" + std::to_string(renderer.msaa_samples()) + " size=" + std::to_string(renderer.width()) + + "x" + std::to_string(renderer.height()) + " gpu_skinning=" + std::to_string(gpu_skinning) + + " terrain=" + std::to_string(native_terrain)); + } + if (g_refresh_rate > 0) { + const bool ok = android_perf::request_frame_rate(renderer.sdl_window(), float(g_refresh_rate)); + SDL_Log("frame rate request %d Hz: %s", g_refresh_rate, ok ? "accepted" : "unavailable"); + } + perf_log.set_fps_cap(g_fps_cap); + perf_log.set_refresh_rate_source([] { return android_perf::display_refresh_rate(); }); + perf_log.set_render_rate_source([] { return android_perf::render_rate(); }); + perf_log.set_battery_source([] { return android_perf::battery_celsius(); }); + // ADPF: each frame's CPU work (everything but the vsync/fence waits and the cap's sleep) against the + // frame budget. Opened once the script thread exists, i.e. here. + using steady = std::chrono::steady_clock; + // The target is 80% of the frame period: the governor settles the clocks so the reported work just + // meets the target, so a target equal to the period leaves every other frame late (cap 60 on the + // test phone: 54 fps against the period, 60 against 80% of it). + const auto frame_budget = [] { + const int fps = g_fps_cap > 0 ? g_fps_cap : (g_refresh_rate > 0 ? g_refresh_rate : 60); + return std::chrono::nanoseconds(800'000'000LL / fps); + }(); + android_perf::PerformanceHint hint; + { + std::vector tids{android_perf::current_thread_id()}; + if (const int script = PythonBoot::ScriptThreadId()) tids.push_back(script); + const bool ok = hint.open(tids, frame_budget.count()); + SDL_Log("ADPF performance hint (%zu threads, %.2f ms): %s", tids.size(), frame_budget.count() / 1e6, + ok ? "on" : "unavailable"); + } + const auto start = steady::now(); double update_ms = 0.0; int completed = 0; std::vector frame_ms; if (frames > 0) frame_ms.reserve(static_cast(frames)); + const auto cap_period = g_fps_cap > 0 ? std::chrono::nanoseconds(1'000'000'000LL / g_fps_cap) + : std::chrono::nanoseconds(0); + auto next_frame_at = steady::now(); while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) { - const auto frame_start = std::chrono::steady_clock::now(); - const auto t0 = std::chrono::steady_clock::now(); + double pace_ms = 0.0; + if (cap_period.count()) { + // Fixed cadence: a late frame starts the next one at once, a frame more than one period + // late re-anchors instead of bursting to catch up. + const auto now = steady::now(); + if (next_frame_at > now) { + std::this_thread::sleep_until(next_frame_at); + pace_ms = std::chrono::duration(steady::now() - now).count(); + } else if (now - next_frame_at > cap_period) { + next_frame_at = now; + } + next_frame_at += cap_period; + } + const auto frame_start = steady::now(); + const auto t0 = frame_start; PythonBoot::UIUpdate(); + const auto t_app = std::chrono::steady_clock::now(); renderer.sync_game_cursor(); PythonBoot::UIRender(); audio.pump(); @@ -3637,9 +4135,20 @@ int run_live_client( unsigned ui_w = renderer.width(), ui_h = renderer.height(); UIRenderGetSize(&ui_w, &ui_h); renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); - const double elapsed_ms = std::chrono::duration( - std::chrono::steady_clock::now() - frame_start).count(); + const auto t2 = std::chrono::steady_clock::now(); + const double elapsed_ms = std::chrono::duration(t2 - frame_start).count(); PythonBoot::SetNativeRenderFrameTime(static_cast(elapsed_ms)); + hint.report(std::int64_t((elapsed_ms - renderer.last_sync_ms()) * 1e6)); + if (perf_log.enabled()) { + native_perf::FrameInput in; + in.frame_ms = elapsed_ms; + in.pace_ms = pace_ms; + in.app_ms = std::chrono::duration(t_app - t0).count(); + in.host_ms = std::chrono::duration(t1 - t_app).count(); + in.render_ms = std::chrono::duration(t2 - t1).count(); + in.renderer = renderer.perf_totals(); + perf_log.frame(in, [] { return PythonBoot::CurrentMapName(); }); + } frame_ms.push_back(elapsed_ms); ++completed; } @@ -3688,6 +4197,7 @@ int main(int argc, char** argv) { bool mobile_mode = false; int safe_inset_px = 0; bool show_fps = false; + int frames_in_flight = 2; #ifdef __ANDROID__ mobile_mode = true; #endif @@ -3742,14 +4252,25 @@ int main(int argc, char** argv) { else if (arg == "--show-fps") show_fps = true; #ifdef MT_NATIVE_HAS_LIVE_CLIENT else if (arg == "--py-exec" && i+1 < argc) g_py_exec_after_login = argv[++i]; + else if (arg == "--perf-log") g_perf_log = true; + else if (arg == "--fps-cap" && i+1 < argc) g_fps_cap = std::max(0, std::stoi(argv[++i])); + else if (arg == "--refresh-rate" && i+1 < argc) g_refresh_rate = std::max(0, std::stoi(argv[++i])); #endif + else if (arg == "--fake-idle-ms" && i+1 < argc) setenv("MT_FAKE_IDLE_MS", argv[++i], 1); + else if (arg == "--frames-in-flight" && i+1 < argc) frames_in_flight = std::stoi(argv[++i]); + else if (arg == "--msaa" && i+1 < argc) setenv("MT_MSAA", argv[++i], 1); + else if (arg == "--cpu-shadow") setenv("MT_CPU_SHADOW", "1", 1); else throw std::runtime_error( "usage: mt_native_render [--frames N] [--interactive] [--login-screen|--selection-screen] " "[--live-server HOST:AUTH_PORT:GAME_PORT] [--width W] [--height H] " "[--draws N] [--triangles-per-draw N] [--capture FILE] [--animate-first-draw] " "[--animate-bones] [--no-vsync] [--live-client DIR] [--fake-mobs N] " - "[--gpu-skinning|--no-gpu-skinning] [--no-terrain] [--mobile] [--capture-out FILE] [--screenshot-out FILE.bmp]"); + "[--gpu-skinning|--no-gpu-skinning] [--no-terrain] [--mobile] [--capture-out FILE] [--screenshot-out FILE.bmp] [--show-fps] [--perf-log] " + "[--fps-cap N] [--refresh-rate HZ] [--frames-in-flight 1|2] [--msaa 1|2|4|8] [--fake-idle-ms MS] [--cpu-shadow]"); } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (native_perf::PerfLog::requested_by_env()) g_perf_log = true; +#endif if (live_client_dir.empty() && !synthetic_specified) { const char* env_client = std::getenv("MT_40250_CLIENT"); if (env_client && *env_client && access((std::string(env_client) + "/pack/Index").c_str(), R_OK) == 0) { @@ -3789,6 +4310,7 @@ int main(int argc, char** argv) { VulkanWindow renderer(vsync, init_width, init_height); renderer.set_safe_inset_px(safe_inset_px); renderer.set_show_fps(show_fps); + renderer.set_frames_in_flight(std::uint32_t(std::max(1, frames_in_flight))); if (!screenshot_out.empty() && frames > 0) renderer.request_screenshot(screenshot_out, std::uint64_t(frames)); if (mobile_mode) { renderer.touch_controller.set_enabled(true); diff --git a/native_render/perf_log.h b/native_render/perf_log.h new file mode 100644 index 00000000..9ef86648 --- /dev/null +++ b/native_render/perf_log.h @@ -0,0 +1,419 @@ +#pragma once +// Performance telemetry for the live client (--perf-log or MT_PERF_LOG=1). +// +// Every frame the host hands PerfLog the frame's wall times and the renderer's running totals; the +// script thread's section times come from platform/EterBase/PerfCounters.h. Every interval (2 s) one CSV +// row with the window's per-frame averages is appended to perf-YYYYMMDD-HHMMSS.csv and a short summary +// goes to logcat (stderr on desktop). A frame over the hitch threshold additionally writes a "# hitch" +// comment line with that frame's own breakdown. +// +// Files: Android /perf (/sdcard/Android/data//files/perf, pulled with +// tools/pull_perf_logs.sh), desktop $MT_PERF_DIR or ./perf. The newest kKeepFiles files are kept. + +#include "platform/EterBase/PerfCounters.h" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#if defined(__linux__) || defined(__ANDROID__) +#include +#include +#endif +#ifdef __ANDROID__ +#include +#endif + +namespace native_perf { + +// Running totals the renderer keeps (VulkanWindow::Timings and its upload counters); PerfLog differences +// them itself, so they never need resetting. +struct RendererTotals { + double sync_ms = 0, fence_ms = 0, prepare_ms = 0, geometry_upload_ms = 0, texture_upload_ms = 0; + double submit_ms = 0, present_ms = 0, gpu_ms = 0; + std::uint64_t gpu_samples = 0; + std::uint64_t uploads = 0, uploaded_bytes = 0, texture_uploads = 0, texture_bytes = 0; + // The last frame's counts. + std::uint64_t draws = 0, skinned_draws = 0, ui_batches = 0, vertices = 0; + const std::string* last_texture = nullptr; // name of the texture uploaded last +}; + +struct FrameInput { + double frame_ms = 0; // whole loop iteration + double app_ms = 0; // PythonBoot::UIUpdate: the script thread's Process() plus the baton handoff + double host_ms = 0; // cursor sync, UIRender, audio pump + double render_ms = 0; // VulkanWindow::render + double pace_ms = 0; // sleep of the frame-rate cap (--fps-cap), outside frame_ms + RendererTotals renderer; +}; + +class PerfLog { +public: + static constexpr double kIntervalS = 2.0; + static constexpr double kHitchMs = 100.0; + static constexpr int kKeepFiles = 10; + + static bool requested_by_env() { + const char* env = std::getenv("MT_PERF_LOG"); + return env && *env && std::string(env) != "0"; + } + + bool enabled() const { return file_ != nullptr; } + // Display refresh rate in Hz as the OS reports it now (Android Display.getRefreshRate), sampled + // once per row; -1 when unknown. + void set_refresh_rate_source(std::function source) { refresh_rate_ = std::move(source); } + // The rate the system actually drives the app at (Android Choreographer); can be a divisor of refresh_hz. + void set_render_rate_source(std::function source) { render_rate_ = std::move(source); } + // Battery temperature in C (Android: the ACTION_BATTERY_CHANGED sticky intent; sysfs is SELinux-denied). + void set_battery_source(std::function source) { battery_ = std::move(source); } + void set_fps_cap(int cap) { fps_cap_ = cap; } + const std::string& path() const { return path_; } + + // `context` goes into the file header (device, present mode, MSAA, size...). + bool open(const std::string& context) { + const std::string dir = directory(); + std::error_code ec; + std::filesystem::create_directories(dir, ec); + prune(dir); + char stamp[32]; + const std::time_t now = std::time(nullptr); + std::tm local{}; + localtime_r(&now, &local); + std::strftime(stamp, sizeof(stamp), "%Y%m%d-%H%M%S", &local); + path_ = dir + "/perf-" + stamp + ".csv"; + file_ = std::fopen(path_.c_str(), "w"); + if (!file_) { + log("perf log: cannot write %s", path_.c_str()); + return false; + } + std::fprintf(file_, "# mt_native_render perf log %s\n# %s\n", stamp, context.c_str()); + std::fprintf(file_, "# per-frame averages over each %.0f s window; *_ms game sections run on the script thread " + "inside app_ms; py_ms includes update_game/render_game\n", kIntervalS); + std::fprintf(file_, + "t_s,map,actors,frames,fps,refresh_hz,render_hz,fps_cap,frame_ms,frame_p95_ms,frame_max_ms,jank33,jank50," + "app_ms,process_ms,handoff_ms,net_ms,camera_ms,ui_update_ms,update_game_ms,render_begin_ms,ui_render_ms," + "render_game_ms,shadow_raster_ms,py_ms,py_calls,py_missing,host_ms,render_ms,pace_ms,sync_ms,fence_ms," + "prepare_ms,geo_upload_ms,tex_upload_ms," + "submit_ms,present_ms,gpu_ms,draws,skinned_draws,ui_batches,vertices,geo_uploads,geo_kb,tex_uploads,tex_kb," + "script_cpu,main_cpu,cpu_mhz,cpu_max_mhz,thermal,batt_c,rss_mb,last_texture\n"); + std::fflush(file_); + MtPerf::Enabled() = true; + MtPerf::Take(); + start_ = window_start_ = std::chrono::steady_clock::now(); + log("perf log: %s", path_.c_str()); + return true; + } + + ~PerfLog() { + if (file_) std::fclose(file_); + MtPerf::Enabled() = false; + } + + // Call once per frame. `map_name` is only asked for when a row is written. + template + void frame(const FrameInput& in, MapName&& map_name) { + if (!file_) return; + const MtPerf::SCounters game = MtPerf::Take(); + const RendererTotals& r = in.renderer; + const RendererTotals& p = have_prev_ ? prev_ : r; + Window f; + f.frames = 1; + f.frame_ms = in.frame_ms; + f.app_ms = in.app_ms; + f.host_ms = in.host_ms; + f.render_ms = in.render_ms; + f.pace_ms = in.pace_ms; + for (int i = 0; i < MtPerf::SECTION_COUNT; ++i) f.game_ms[i] = game.ms[i]; + f.py_calls = game.calls[MtPerf::SECTION_PYTHON]; + f.py_missing = game.pyMissing; + f.sync_ms = r.sync_ms - p.sync_ms; + f.fence_ms = r.fence_ms - p.fence_ms; + f.prepare_ms = r.prepare_ms - p.prepare_ms; + f.geo_upload_ms = r.geometry_upload_ms - p.geometry_upload_ms; + f.tex_upload_ms = r.texture_upload_ms - p.texture_upload_ms; + f.submit_ms = r.submit_ms - p.submit_ms; + f.present_ms = r.present_ms - p.present_ms; + f.gpu_ms = r.gpu_ms - p.gpu_ms; + f.gpu_samples = r.gpu_samples - p.gpu_samples; + f.geo_uploads = r.uploads - p.uploads; + f.geo_bytes = r.uploaded_bytes - p.uploaded_bytes; + f.tex_uploads = r.texture_uploads - p.texture_uploads; + f.tex_bytes = r.texture_bytes - p.texture_bytes; + prev_ = r; + have_prev_ = true; + + if (in.frame_ms >= kHitchMs) write_hitch(f, game); + window_.add(f); + frame_times_.push_back(in.frame_ms); + draws_ = r.draws; + skinned_draws_ = r.skinned_draws; + ui_batches_ = r.ui_batches; + vertices_ = r.vertices; + if (f.tex_uploads && r.last_texture) last_texture_ = *r.last_texture; + actors_ = game.actorCount; + script_cpu_ = game.scriptCpu; + + const auto now = std::chrono::steady_clock::now(); + const double elapsed = std::chrono::duration(now - window_start_).count(); + if (elapsed < kIntervalS) return; + write_row(elapsed, map_name()); + window_ = {}; + frame_times_.clear(); + window_start_ = now; + } + +private: + struct Window { + std::uint64_t frames = 0; + double frame_ms = 0, app_ms = 0, host_ms = 0, render_ms = 0, pace_ms = 0; + double game_ms[MtPerf::SECTION_COUNT] = {}; + std::uint64_t py_calls = 0, py_missing = 0; + double sync_ms = 0, fence_ms = 0, prepare_ms = 0, geo_upload_ms = 0, tex_upload_ms = 0, submit_ms = 0; + double present_ms = 0, gpu_ms = 0; + std::uint64_t gpu_samples = 0; + std::uint64_t geo_uploads = 0, geo_bytes = 0, tex_uploads = 0, tex_bytes = 0; + + void add(const Window& o) { + frames += o.frames; + frame_ms += o.frame_ms; app_ms += o.app_ms; host_ms += o.host_ms; render_ms += o.render_ms; + pace_ms += o.pace_ms; fence_ms += o.fence_ms; + for (int i = 0; i < MtPerf::SECTION_COUNT; ++i) game_ms[i] += o.game_ms[i]; + py_calls += o.py_calls; py_missing += o.py_missing; + sync_ms += o.sync_ms; prepare_ms += o.prepare_ms; geo_upload_ms += o.geo_upload_ms; + tex_upload_ms += o.tex_upload_ms; submit_ms += o.submit_ms; present_ms += o.present_ms; + gpu_ms += o.gpu_ms; gpu_samples += o.gpu_samples; + geo_uploads += o.geo_uploads; geo_bytes += o.geo_bytes; + tex_uploads += o.tex_uploads; tex_bytes += o.tex_bytes; + } + }; + + static std::string directory() { + if (const char* env = std::getenv("MT_PERF_DIR"); env && *env) return env; +#ifdef __ANDROID__ + if (const char* ext = SDL_GetAndroidExternalStoragePath(); ext && *ext) return std::string(ext) + "/perf"; + if (const char* internal = SDL_GetAndroidInternalStoragePath(); internal && *internal) + return std::string(internal) + "/perf"; +#endif + return "perf"; + } + + static void prune(const std::string& dir) { + std::error_code ec; + std::vector files; + for (const auto& entry : std::filesystem::directory_iterator(dir, ec)) { + const std::string name = entry.path().filename().string(); + if (name.rfind("perf-", 0) == 0 && entry.path().extension() == ".csv") files.push_back(entry.path()); + } + std::sort(files.begin(), files.end()); + // Leave room for the file about to be opened. + for (std::size_t i = 0; i + kKeepFiles <= files.size(); ++i) std::filesystem::remove(files[i], ec); + } + + template + static void log(const char* format, Args... args) { +#ifdef __ANDROID__ + SDL_Log(format, args...); +#else + std::fprintf(stderr, format, args...); + std::fputc('\n', stderr); +#endif + } + + static int current_cpu() { +#if defined(__linux__) || defined(__ANDROID__) + return sched_getcpu(); +#else + return -1; +#endif + } + + static long read_long(const char* path) { + std::FILE* f = std::fopen(path, "r"); + if (!f) return -1; + long value = -1; + if (std::fscanf(f, "%ld", &value) != 1) value = -1; + std::fclose(f); + return value; + } + + // Current clock of every CPU in MHz, "cpu0/cpu1/...", or "" where sysfs is closed. + static std::string cpu_mhz() { + std::string out; +#if defined(__linux__) || defined(__ANDROID__) + for (int cpu = 0; cpu < 16; ++cpu) { + char path[96]; + std::snprintf(path, sizeof(path), "/sys/devices/system/cpu/cpu%d/cpufreq/scaling_cur_freq", cpu); + const long khz = read_long(path); + if (khz < 0) { + if (cpu >= 8 || access(path, F_OK) != 0) break; + continue; + } + if (!out.empty()) out += '/'; + out += std::to_string(khz / 1000); + } +#endif + return out; + } + + // Each cpufreq policy's current ceiling (scaling_max_freq): the vendor's thermal throttling shows up + // here long before AThermal reports anything. + static std::string cpu_max_mhz() { + std::string out; +#if defined(__linux__) || defined(__ANDROID__) + for (int policy = 0; policy < 16; ++policy) { + char path[96]; + std::snprintf(path, sizeof(path), "/sys/devices/system/cpu/cpufreq/policy%d/scaling_max_freq", policy); + const long khz = read_long(path); + if (khz < 0) continue; + if (!out.empty()) out += '/'; + out += std::to_string(khz / 1000); + } +#endif + return out; + } + + // AThermal_getCurrentThermalStatus (API 30+, looked up at run time since minSdk is lower): + // 0 none, 1 light, 2 moderate, 3 severe, 4 critical, 5 emergency, 6 shutdown; -1 unknown. + static int thermal_status() { +#ifdef __ANDROID__ + using Acquire = void* (*)(); + using Status = int (*)(void*); + static void* manager = nullptr; + static Status status = nullptr; + static bool looked_up = false; + if (!looked_up) { + looked_up = true; + if (void* lib = dlopen("libandroid.so", RTLD_NOW)) { + const auto acquire = reinterpret_cast(dlsym(lib, "AThermal_acquireManager")); + status = reinterpret_cast(dlsym(lib, "AThermal_getCurrentThermalStatus")); + if (acquire && status) manager = acquire(); + } + } + if (manager && status) return status(manager); +#endif + return -1; + } + + double battery_celsius() const { + if (battery_) return battery_(); +#if defined(__linux__) || defined(__ANDROID__) + const long tenths = read_long("/sys/class/power_supply/battery/temp"); + if (tenths > 0) return double(tenths) / 10.0; // -1: unreadable (SELinux on most phones) +#endif + return -1.0; + } + + static double rss_mb() { +#if defined(__linux__) || defined(__ANDROID__) + std::FILE* f = std::fopen("/proc/self/statm", "r"); + if (!f) return -1.0; + long pages = 0, resident = 0; + const int read = std::fscanf(f, "%ld %ld", &pages, &resident); + std::fclose(f); + if (read == 2) return double(resident) * double(sysconf(_SC_PAGESIZE)) / (1024.0 * 1024.0); +#endif + return -1.0; + } + + static std::string csv_field(std::string text) { + for (char& c : text) + if (c == ',' || c == '\n' || c == '\r') c = ' '; + return text; + } + + void write_row(double elapsed_s, const std::string& map) { + const Window& w = window_; + if (!w.frames) return; + const double n = double(w.frames); + std::vector sorted = frame_times_; + std::sort(sorted.begin(), sorted.end()); + const double p95 = sorted[std::min(sorted.size() - 1, std::size_t(0.95 * double(sorted.size())))]; + int jank33 = 0, jank50 = 0; + for (double ms : frame_times_) { + jank33 += ms > 33.4; + jank50 += ms > 50.0; + } + const auto g = [&](MtPerf::ESection s) { return w.game_ms[s] / n; }; + const double fps = n / elapsed_s; + const double gpu = w.gpu_samples ? w.gpu_ms / double(w.gpu_samples) : 0.0; + const double t = std::chrono::duration(std::chrono::steady_clock::now() - start_).count(); + const std::string mhz = cpu_mhz(); + const std::string max_mhz = cpu_max_mhz(); + const int thermal = thermal_status(); + const double batt = battery_celsius(); + const double rss = rss_mb(); + const int main_cpu = current_cpu(); + const double refresh = refresh_rate_ ? refresh_rate_() : -1.0; + const double render_rate = render_rate_ ? render_rate_() : -1.0; + std::fprintf(file_, + "%.1f,%s,%d,%llu,%.1f,%.1f,%.1f,%d,%.2f,%.2f,%.2f,%d,%d," + "%.2f,%.2f,%.2f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f," + "%.3f,%.3f,%.3f,%.1f,%.1f,%.3f,%.2f,%.2f,%.2f,%.2f," + "%.2f,%.3f,%.3f," + "%.3f,%.3f,%.2f,%llu,%llu,%llu,%llu,%.2f,%.1f,%.2f,%.1f," + "%d,%d,%s,%s,%d,%.1f,%.0f,%s\n", + t, csv_field(map).c_str(), actors_, (unsigned long long)w.frames, fps, refresh, render_rate, fps_cap_, w.frame_ms / n, + p95, sorted.back(), jank33, jank50, + w.app_ms / n, g(MtPerf::SECTION_PROCESS), (w.app_ms - w.game_ms[MtPerf::SECTION_PROCESS]) / n, + g(MtPerf::SECTION_NETWORK), g(MtPerf::SECTION_CAMERA), g(MtPerf::SECTION_UI_UPDATE), + g(MtPerf::SECTION_UPDATE_GAME), g(MtPerf::SECTION_RENDER_BEGIN), g(MtPerf::SECTION_UI_RENDER), + g(MtPerf::SECTION_RENDER_GAME), g(MtPerf::SECTION_SHADOW_RASTER), g(MtPerf::SECTION_PYTHON), + double(w.py_calls) / n, + double(w.py_missing) / n, w.host_ms / n, w.render_ms / n, w.pace_ms / n, w.sync_ms / n, w.fence_ms / n, + w.prepare_ms / n, w.geo_upload_ms / n, w.tex_upload_ms / n, + w.submit_ms / n, w.present_ms / n, gpu, (unsigned long long)draws_, (unsigned long long)skinned_draws_, + (unsigned long long)ui_batches_, (unsigned long long)vertices_, double(w.geo_uploads) / n, + double(w.geo_bytes) / 1024.0 / n, double(w.tex_uploads) / n, double(w.tex_bytes) / 1024.0 / n, + script_cpu_, main_cpu, mhz.c_str(), max_mhz.c_str(), thermal, batt, rss, csv_field(last_texture_).c_str()); + std::fflush(file_); + if (!w.tex_uploads) last_texture_.clear(); + log("perf fps=%.1f @%.0fHz/%.0f cap=%d frame=%.1f/p95 %.1f/max %.1f jank33=%d | app=%.1f (upd=%.1f rnd=%.1f " + "shadow=%.1f py=%.1f) | render=%.1f (sync=%.1f fence=%.1f prep=%.1f up=%.1f/%.1f) pace=%.1f gpu=%.1f | " + "draws=%llu actors=%d map=%s thermal=%d batt=%.1f cpu=%s max=%s tex_uploads=%.2f last_tex=%s", + fps, refresh, render_rate, fps_cap_, w.frame_ms / n, p95, sorted.back(), jank33, w.app_ms / n, + g(MtPerf::SECTION_UPDATE_GAME), g(MtPerf::SECTION_RENDER_GAME), g(MtPerf::SECTION_SHADOW_RASTER), + g(MtPerf::SECTION_PYTHON), w.render_ms / n, w.sync_ms / n, w.fence_ms / n, w.prepare_ms / n, + w.geo_upload_ms / n, w.tex_upload_ms / n, w.pace_ms / n, gpu, (unsigned long long)draws_, actors_, + map.c_str(), thermal, batt, mhz.c_str(), max_mhz.c_str(), double(w.tex_uploads) / n, last_texture_.c_str()); + } + + void write_hitch(const Window& f, const MtPerf::SCounters& game) { + const double t = std::chrono::duration(std::chrono::steady_clock::now() - start_).count(); + std::fprintf(file_, + "# hitch t=%.2f frame=%.1f app=%.1f process=%.1f net=%.1f camera=%.1f ui_update=%.1f update_game=%.1f " + "ui_render=%.1f render_game=%.1f shadow=%.1f py=%.1f host=%.1f render=%.1f sync=%.1f prepare=%.1f geo_upload=%.1f(%llu) " + "tex_upload=%.1f(%llu, %.0f KB) submit=%.1f present=%.1f\n", + t, f.frame_ms, f.app_ms, game.ms[MtPerf::SECTION_PROCESS], game.ms[MtPerf::SECTION_NETWORK], + game.ms[MtPerf::SECTION_CAMERA], game.ms[MtPerf::SECTION_UI_UPDATE], game.ms[MtPerf::SECTION_UPDATE_GAME], + game.ms[MtPerf::SECTION_UI_RENDER], game.ms[MtPerf::SECTION_RENDER_GAME], + game.ms[MtPerf::SECTION_SHADOW_RASTER], game.ms[MtPerf::SECTION_PYTHON], + f.host_ms, f.render_ms, f.sync_ms, f.prepare_ms, f.geo_upload_ms, (unsigned long long)f.geo_uploads, + f.tex_upload_ms, (unsigned long long)f.tex_uploads, double(f.tex_bytes) / 1024.0, f.submit_ms, f.present_ms); + } + + std::FILE* file_ = nullptr; + std::string path_; + std::chrono::steady_clock::time_point start_{}, window_start_{}; + Window window_; + std::vector frame_times_; + RendererTotals prev_; + bool have_prev_ = false; + std::uint64_t draws_ = 0, skinned_draws_ = 0, ui_batches_ = 0, vertices_ = 0; + int actors_ = 0, script_cpu_ = -1; + int fps_cap_ = 0; + std::function refresh_rate_; + std::function render_rate_; + std::function battery_; + std::string last_texture_; +}; + +} // namespace native_perf diff --git a/tools/perf_summary.py b/tools/perf_summary.py new file mode 100644 index 00000000..9b2fdd21 --- /dev/null +++ b/tools/perf_summary.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +"""Summarises a native client --perf-log CSV (native_render/perf_log.h). + +usage: perf_summary.py perf-YYYYMMDD-HHMMSS.csv [--all] + +Only in-game rows (a map is loaded) are summarised unless --all is given. Prints the median of each +frame-time section so the largest cost stands out, the worst windows, thermal/clock trend and the +"# hitch" lines. +""" +import csv +import statistics +import sys + +SECTIONS = [ + ("frame_ms", "whole frame"), + ("app_ms", " script thread (AppFrame)"), + ("net_ms", " network"), + ("camera_ms", " camera/resource"), + ("update_game_ms", " UpdateGame"), + ("render_game_ms", " RenderGame"), + ("shadow_raster_ms", " CPU shadow raster"), + ("py_ms", " Python callbacks (incl. Update/RenderGame)"), + ("handoff_ms", " baton handoff"), + ("pace_ms", " fps-cap sleep (before the frame)"), + ("host_ms", " host UI/audio"), + ("render_ms", " renderer"), + ("sync_ms", " fence wait + acquire"), + ("fence_ms", " frame-slot fence wait"), + ("prepare_ms", " prepare (incl. uploads)"), + ("geo_upload_ms", " geometry uploads"), + ("tex_upload_ms", " texture uploads"), + ("submit_ms", " submit"), + ("present_ms", " present"), + ("gpu_ms", " GPU (timestamps)"), +] + + +def num(row, key): + try: + return float(row.get(key, "") or "nan") + except ValueError: + return float("nan") + + +def median(rows, key): + values = [num(r, key) for r in rows] + values = [v for v in values if v == v] + return statistics.median(values) if values else float("nan") + + +def main(): + args = [a for a in sys.argv[1:] if not a.startswith("--")] + if not args: + sys.exit(__doc__) + path = args[0] + comments, lines = [], [] + with open(path, encoding="utf-8", errors="replace") as f: + for line in f: + (comments if line.startswith("#") else lines).append(line) + rows = list(csv.DictReader(lines)) + if "--all" not in sys.argv: + rows = [r for r in rows if r.get("map")] + print(path) + for c in comments[:2]: + print(c.rstrip()) + if not rows: + print("no in-game rows (use --all to include the login/select screens)") + return + duration = num(rows[-1], "t_s") - num(rows[0], "t_s") + print("%d windows, %.0f s in game, maps: %s" % (len(rows), duration, ", ".join(sorted({r["map"] for r in rows})))) + print("display %s Hz app vsync (render_hz) median %.1f fps cap %s" % ( + "/".join(sorted({r.get("refresh_hz", "?") for r in rows})), median(rows, "render_hz"), + rows[-1].get("fps_cap", "?"))) + print("median fps %.1f frame p95 %.1f ms jank>33ms %d frames jank>50ms %d frames" % ( + median(rows, "fps"), median(rows, "frame_p95_ms"), + sum(int(num(r, "jank33")) for r in rows), sum(int(num(r, "jank50")) for r in rows))) + print("median actors %.0f draws %.0f skinned %.0f python calls/frame %.0f (missing %.0f)" % ( + median(rows, "actors"), median(rows, "draws"), median(rows, "skinned_draws"), + median(rows, "py_calls"), median(rows, "py_missing"))) + print("uploads/frame: geometry %.1f (%.1f KB) texture %.2f (%.0f KB)" % ( + median(rows, "geo_uploads"), median(rows, "geo_kb"), median(rows, "tex_uploads"), median(rows, "tex_kb"))) + print() + print("median ms per frame") + for key, label in SECTIONS: + print(" %-48s %7.2f" % (label, median(rows, key))) + + print() + print("worst windows (lowest fps)") + for r in sorted(rows, key=lambda r: num(r, "fps"))[:5]: + print(" t=%6.0fs fps=%5.1f frame=%5.1f app=%5.1f render_game=%5.1f render=%5.1f sync=%5.1f gpu=%5.1f " + "actors=%s thermal=%s cpu_mhz=%s" % ( + num(r, "t_s"), num(r, "fps"), num(r, "frame_ms"), num(r, "app_ms"), num(r, "render_game_ms"), + num(r, "render_ms"), num(r, "sync_ms"), num(r, "gpu_ms"), r.get("actors"), r.get("thermal"), + r.get("cpu_mhz"))) + + step = max(1, len(rows) // 8) + print() + print("trend (thermal: 0 none .. 3 severe; -1 unknown)") + for r in rows[::step]: + print(" t=%6.0fs fps=%5.1f thermal=%s batt=%s C cpu_mhz=%s (max %s) rss=%s MB" % ( + num(r, "t_s"), num(r, "fps"), r.get("thermal"), r.get("batt_c"), r.get("cpu_mhz"), + r.get("cpu_max_mhz", "?"), r.get("rss_mb"))) + + hitches = [c.rstrip() for c in comments if c.startswith("# hitch")] + if hitches: + print() + print("%d hitches >= 100 ms (first 10)" % len(hitches)) + for h in hitches[:10]: + print(" " + h[2:]) + + +if __name__ == "__main__": + main() diff --git a/tools/pull_perf_logs.sh b/tools/pull_perf_logs.sh new file mode 100755 index 00000000..8373c34f --- /dev/null +++ b/tools/pull_perf_logs.sh @@ -0,0 +1,11 @@ +#!/bin/sh +# Copies the Android client's --perf-log files (native_render/perf_log.h) to ./perf-logs and prints a +# summary of the newest one. The phone must be connected with USB debugging; the files can also be +# taken off the phone without a PC from Android/data/org.metin2port.client/files/perf. +set -e +PKG=org.metin2port.client +OUT=${1:-perf-logs} +mkdir -p "$OUT" +adb pull "/sdcard/Android/data/$PKG/files/perf/." "$OUT" +newest=$(ls -t "$OUT"/perf-*.csv 2>/dev/null | head -1) +[ -n "$newest" ] && python3 "$(dirname "$0")/perf_summary.py" "$newest"