From 75d2dd8545d2bc399897c78f15acfbbef9fdc555 Mon Sep 17 00:00:00 2001 From: shenlei Date: Tue, 29 Sep 2026 19:30:44 +0900 Subject: [PATCH] host: split src/host/main.cpp into translation units main.cpp (4.5k lines) -> main.cpp (entry, args, synthetic benchmark) plus vulkan_window, live_client, frame_policy, native_audio, texture_decode, render_state, input_keymap, host_util. Bodies are moved verbatim; multi-line VulkanWindow / NativeAudioEngine / FrameRatePolicy members are defined out of line; one-line accessors stay inline. Anonymous namespace -> mt_host. No behavior change: ctest 15/15, port_gate macos+android PASS, desktop --live-client login screen and synthetic --draws runs unchanged. Co-Authored-By: Claude Opus 5.5 --- .../references/project-map.md | 2 +- .../port-map/EterLib/GrpImageTexture.cpp.json | 6 +- docs/PORT-PLAN.md | 14 +- src/host/CMakeLists.txt | 8 + src/host/android_perf.h | 2 +- src/host/frame_policy.cpp | 64 + src/host/frame_policy.h | 50 + src/host/host_util.cpp | 36 + src/host/host_util.h | 19 + src/host/input_keymap.cpp | 147 + src/host/input_keymap.h | 12 + src/host/live_client.cpp | 514 ++ src/host/live_client.h | 31 + src/host/main.cpp | 4266 +---------------- src/host/native_audio.cpp | 267 ++ src/host/native_audio.h | 67 + src/host/render_state.cpp | 130 + src/host/render_state.h | 75 + src/host/shaders/native.vert | 2 +- src/host/texture_decode.cpp | 154 + src/host/texture_decode.h | 18 + src/host/vulkan_window.cpp | 2679 +++++++++++ src/host/vulkan_window.h | 437 ++ 23 files changed, 4729 insertions(+), 4271 deletions(-) create mode 100644 src/host/frame_policy.cpp create mode 100644 src/host/frame_policy.h create mode 100644 src/host/host_util.cpp create mode 100644 src/host/host_util.h create mode 100644 src/host/input_keymap.cpp create mode 100644 src/host/input_keymap.h create mode 100644 src/host/live_client.cpp create mode 100644 src/host/live_client.h create mode 100644 src/host/native_audio.cpp create mode 100644 src/host/native_audio.h create mode 100644 src/host/render_state.cpp create mode 100644 src/host/render_state.h create mode 100644 src/host/texture_decode.cpp create mode 100644 src/host/texture_decode.h create mode 100644 src/host/vulkan_window.cpp create mode 100644 src/host/vulkan_window.h diff --git a/.agents/skills/metin2-40250-parity-audit/references/project-map.md b/.agents/skills/metin2-40250-parity-audit/references/project-map.md index a8d9ccc4..0acd48f6 100644 --- a/.agents/skills/metin2-40250-parity-audit/references/project-map.md +++ b/.agents/skills/metin2-40250-parity-audit/references/project-map.md @@ -33,7 +33,7 @@ - `src/port//.{h,cpp}` — mirror of each 40250 `logic`/`python` unit (see SKILL.md "Code layout"). - `src/platform/` — implementations of the mirrored 40250 platform classes (D3D8 recorder, Granny, Miles, SpeedTree stand-in, window/input, ScriptLib boot). -- `src/host/` — SDL3 + Vulkan host (`main.cpp`); `src/gr2/` — Granny reader; `src/codepage/` — code pages. +- `src/host/` — SDL3 + Vulkan host (`main.cpp` entry, `live_client.cpp` loop, `vulkan_window.cpp` renderer); `src/gr2/` — Granny reader; `src/codepage/` — code pages. - `third_party/` — vendored cpython 2.7.18, cryptopp, freetype, minilzo. - `tests/port/` — port tests and the fake login server; `tests/gr2/` — Granny reader tests. - `android/` — SDLActivity shell, `android/build.sh`, `android/push-client.sh`. diff --git a/audit/port-map/EterLib/GrpImageTexture.cpp.json b/audit/port-map/EterLib/GrpImageTexture.cpp.json index e55dca17..2633870f 100644 --- a/audit/port-map/EterLib/GrpImageTexture.cpp.json +++ b/audit/port-map/EterLib/GrpImageTexture.cpp.json @@ -29,8 +29,8 @@ "status": "ADAPTED", "impl": [ "src/host/dxt.cpp:decode", - "src/host/main.cpp:get_gpu_texture", - "src/host/main.cpp:upload_texture" + "src/host/vulkan_window.cpp:get_gpu_texture", + "src/host/vulkan_window.cpp:upload_texture" ], "test": "tests/port/dxt_decode_test.cpp", "note": "The platform CreateDDSTexture member is a stub; its level logic lives in the texture decode/upload the renderer does. mipmapCount = max(1, min(dwMipMapCount, MAX_MIPLEVELS=12)), level i read at sum(dwLinearSize>>2k) with CDXTCImage::Copy's dwLinearSize>>2i bytes (the rest zero, uninitialised in 40250), decoded to RGBA8 (no GPU DXT) and uploaded as file levels, not regenerated. Divergence: without DDSD_MIPMAPCOUNT, or with DDSD_PITCH, 40250 creates the extra levels uninitialised; here that is one level. IsLowTextureMemory's uTexBias level skip is not applied (the port's ms_isLowTextureMemory is false). The far-terrain look NEEDS_LIVE." @@ -39,7 +39,7 @@ "status": "ADAPTED", "impl": [ "src/platform/EterLib/GrpImageTexture.cpp:CGraphicImageTexture::CreateFromMemoryFile", - "src/host/main.cpp:get_gpu_texture" + "src/host/vulkan_window.cpp:get_gpu_texture" ], "test": "tests/port/dxt_decode_test.cpp", "note": "The platform body records only the size and file name; the renderer decodes the bytes. 40250 dispatch is kept: a DXT DDS takes the CreateDDSTexture levels, anything else takes D3DXCreateTextureFromFileInMemoryEx(MipLevels=D3DX_DEFAULT), i.e. a full generated chain (vkCmdBlitImage linear, instead of D3DX's box filter). The invented is_ui_tex rule (no mips for icon/, ui/, locale/ and textures <=128px) was removed. Memory textures (mem:, 40250 CreateTexture levels=1) get one level. GRAPHICS_CAPS_HALF_SIZE_IMAGE/low-memory A4R4G4B4 re-encode is not applied." diff --git a/docs/PORT-PLAN.md b/docs/PORT-PLAN.md index 5067f939..06ecc44f 100644 --- a/docs/PORT-PLAN.md +++ b/docs/PORT-PLAN.md @@ -62,15 +62,19 @@ src/platform/ # 这些平台类的实现(CGraph src/platform/EterLib/RecordingDevice.cpp # IDirect3DDevice8 的记录实现:把 D3D8 调用录成绘制命令 src/platform/ScriptLib/PythonBoot.cpp # 解释器启动、RunMainScript、脚本线程、PORT 模块注入 third_party/cpython-2.7.18/ # 静态库 -src/host/main.cpp # SDL3 宿主:窗口/输入/触控、VulkanWindow 渲染器、帧率策略 -src/host/android_perf.h # Android 帧率、ADPF、电量/温度(JNI 到 MainActivity) -android/ # Gradle 工程:SDLActivity 外壳 MainActivity、APK 打包 +src/host/main.cpp # SDL3 宿主入口:命令行、合成基准 +src/host/live_client.cpp # run_live_client:40250 客户端主循环 +src/host/vulkan_window.cpp # SDL 窗口/事件 + Vulkan 渲染器(回放 Render3DDraw / UIRenderCommand) +src/host/frame_policy.cpp # 帧率策略(FrameRatePolicy) +src/host/native_audio.cpp # MilesLib AudioCommand → SDL 音频流 +src/host/android_perf.h # Android 帧率、ADPF、电量/温度(JNI 到 MainActivity) +android/ # Gradle 工程:SDLActivity 外壳 MainActivity、APK 打包 ``` 运行结构(每帧): ``` -SDL3 宿主主循环(src/host/main.cpp run_live_client) +SDL3 宿主主循环(src/host/live_client.cpp run_live_client) ├─ SDL 事件 → 40250 的窗口消息/输入接口(鼠标、键盘、IME 文本、触控摇杆/按钮) ├─ PythonBoot::AppFrame():交给脚本线程跑一次 CPythonApplication::Process()(宿主等待,两者不并发) │ └─ system.py / app.Loop() → UpdateGame / RenderGame / 窗口系统 Render @@ -108,7 +112,7 @@ PORT 扩展(40250 没有对应物,只为移动端/平台而加;不能改 4 - 由 `PythonBoot::RunMainScript` 注入的 Python 模块:`mt_autohunt`(自动狩猎,`AutoHuntScript.inc`)、`mt_register` (客户端内账号注册)、`mt_framerate`(系统选项里的"画面帧率"行)。它们包装 40250 窗口类的方法,不修改脚本文件。 - 触控控制(`src/host/touch_controller.h`):摇杆、攻击/技能按钮,转成 40250 的输入接口。 -- 帧率与功耗(`platform/EterBase/FrameRateMode.h` + `main.cpp` 的 `FrameRatePolicy`):30 / 60(默认)/ 显示器最高三档, +- 帧率与功耗(`platform/EterBase/FrameRateMode.h` + `src/host/frame_policy.cpp` 的 `FrameRatePolicy`):30 / 60(默认)/ 显示器最高三档, 无输入一段时间降到 30 fps(自动狩猎时除外),Android 上配合 `setFrameRate` 和 ADPF。 - 性能遥测:`--perf-log` 写 CSV(`src/host/perf_log.h`、`platform/EterBase/PerfCounters.h`)。 diff --git a/src/host/CMakeLists.txt b/src/host/CMakeLists.txt index 7c32fafe..af73aecb 100644 --- a/src/host/CMakeLists.txt +++ b/src/host/CMakeLists.txt @@ -14,6 +14,14 @@ add_custom_target(mt_native_shaders DEPENDS "${MT_NATIVE_VERT_SPV}" "${MT_NATIVE set(MT_NATIVE_RENDER_SOURCES main.cpp + frame_policy.cpp + host_util.cpp + input_keymap.cpp + live_client.cpp + native_audio.cpp + render_state.cpp + texture_decode.cpp + vulkan_window.cpp stb_image_impl.cpp dxt.cpp ) diff --git a/src/host/android_perf.h b/src/host/android_perf.h index 7fc99583..1b9a1c67 100644 --- a/src/host/android_perf.h +++ b/src/host/android_perf.h @@ -11,7 +11,7 @@ // frame budget, so the governor raises the clocks before a frame misses instead of idling the CPU at // its lowest step between short bursts (the 556-748 MHz seen in perf-20260929-153205.csv). // prefer_power_efficiency(): APerformanceHint_setPreferPowerEfficiency (API 35), set by the frame-rate -// policy (main.cpp FrameRatePolicy) except in the highest mode. +// policy (frame_policy.cpp FrameRatePolicy) except in the highest mode. // - set_display_refresh_rate() / max_refresh_rate(): MainActivity switches the display mode at run time // for the frame-rate setting (60 Hz unless the setting wants more). // - battery_power(): BatteryManager current + voltage for the perf log's power columns; the power is only diff --git a/src/host/frame_policy.cpp b/src/host/frame_policy.cpp new file mode 100644 index 00000000..99886de6 --- /dev/null +++ b/src/host/frame_policy.cpp @@ -0,0 +1,64 @@ +#include "frame_policy.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "android_perf.h" +#include "platform/EterBase/FrameRateMode.h" + +#include + +#include +#include +#include + +namespace mt_host { + +// --fps-cap N: the loop starts a frame at most every 1/N s (0 = as fast as vsync allows). +int g_fps_cap = 0; +// --refresh-rate N: the surface's preferred frame rate (Android ANativeWindow_setFrameRate; MainActivity +// also picks the matching display mode). 0 leaves the system default. +int g_refresh_rate = 0; +// --fps-mode N: the frame-rate setting for this run (a test override: the saved setting stays); +// --idle-fps N: the rate after --idle-after S seconds without input (0 = never lower it). +int g_fps_mode = -1; +int g_idle_fps = 30; +double g_idle_after_s = 15.0; + +void FrameRatePolicy::load() { + path_ = settings_path(); + int mode = MtFrameRate::MODE_STANDARD; + if (std::ifstream in{path_}; in) in >> mode; + if (g_fps_mode >= 0) mode = g_fps_mode; + MtFrameRate::Set(mode); + saved_mode_ = MtFrameRate::Get(); + max_hz_ = int(std::lround(android_perf::max_refresh_rate())); + if (max_hz_ < 30) max_hz_ = 60; +} + +FrameRatePolicy::Decision FrameRatePolicy::decide(double idle_seconds, bool auto_hunt) { + Decision d; + d.mode = MtFrameRate::Get(); + if (d.mode != saved_mode_) save(saved_mode_ = d.mode); + const int active = d.mode == MtFrameRate::MODE_SAVER ? 30 + : d.mode == MtFrameRate::MODE_HIGH ? max_hz_ : std::min(60, max_hz_); + d.idle = g_idle_fps > 0 && g_idle_fps < active && idle_seconds >= g_idle_after_s && !auto_hunt; + d.fps = d.idle ? g_idle_fps : active; + d.display_hz = d.mode == MtFrameRate::MODE_HIGH && !d.idle ? max_hz_ : std::min(60, max_hz_); + return d; +} + +std::string FrameRatePolicy::settings_path() { + std::string dir; + if (char* pref = SDL_GetPrefPath("metin2port", "client")) { + dir = pref; + SDL_free(pref); + } + return dir + "frame_rate.cfg"; +} + +void FrameRatePolicy::save(int mode) { + std::ofstream out{path_, std::ios::trunc}; + out << mode << '\n'; +} + +} // namespace mt_host +#endif diff --git a/src/host/frame_policy.h b/src/host/frame_policy.h new file mode 100644 index 00000000..fbb8fe9a --- /dev/null +++ b/src/host/frame_policy.h @@ -0,0 +1,50 @@ +#pragma once + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include + +namespace mt_host { + +// Command-line overrides, defined in frame_policy.cpp. +extern int g_fps_cap; +extern int g_refresh_rate; +extern int g_fps_mode; +extern int g_idle_fps; +extern double g_idle_after_s; + +// The frame-rate setting (system option dialog, platform/EterBase/FrameRateMode.h) turned into a frame +// rate, a display refresh rate and the ADPF budget; --fps-cap / --refresh-rate pin those for a test run. +// Frames the player does not watch closely cost the same power as the ones they do, so after a while +// without input (and no auto hunt running) the rate drops to --idle-fps until the next touch. The display +// is asked for 60 Hz unless the setting wants more: a 120 Hz panel scanning out a 60 fps game costs power +// for nothing. +class FrameRatePolicy { +public: + struct Decision { + int mode = -1; + int fps = 0; // frames per second the loop aims at + int display_hz = 0; // the refresh rate asked of the display + bool idle = false; + bool operator==(const Decision&) const = default; + }; + + bool pinned() const { return g_fps_cap > 0 || g_refresh_rate > 0; } + + void load(); + + Decision decide(double idle_seconds, bool auto_hunt); + + int max_hz() const { return max_hz_; } + +private: + static std::string settings_path(); + + void save(int mode); + + std::string path_; + int saved_mode_ = -1; + int max_hz_ = 60; +}; + +} // namespace mt_host +#endif diff --git a/src/host/host_util.cpp b/src/host/host_util.cpp new file mode 100644 index 00000000..1228b282 --- /dev/null +++ b/src/host/host_util.cpp @@ -0,0 +1,36 @@ +#include "host_util.h" + +#include +#ifdef __ANDROID__ +#include +#endif + +#include + +namespace mt_host { + +void check(VkResult result, const char* operation) { + if (result != VK_SUCCESS) throw std::runtime_error(std::string(operation) + ": VkResult " + std::to_string(result)); +} + +std::string bundled_file_path(const char* name) { + const char* base = SDL_GetBasePath(); + return std::string(base ? base : "./") + name; +} + +#ifdef __ANDROID__ +// MainActivity.performHaptic(): the long-press vibration of the touch gestures. +void android_haptic() { + auto* env = static_cast(SDL_GetAndroidJNIEnv()); + auto activity = static_cast(SDL_GetAndroidActivity()); + if (!env || !activity) return; + jclass cls = env->GetObjectClass(activity); + if (jmethodID method = env->GetMethodID(cls, "performHaptic", "()V")) + env->CallVoidMethod(activity, method); + if (env->ExceptionCheck()) env->ExceptionClear(); + env->DeleteLocalRef(cls); + env->DeleteLocalRef(activity); +} +#endif + +} // namespace mt_host diff --git a/src/host/host_util.h b/src/host/host_util.h new file mode 100644 index 00000000..1a4ea675 --- /dev/null +++ b/src/host/host_util.h @@ -0,0 +1,19 @@ +#pragma once + +#include + +#include + +namespace mt_host { + +// Throws std::runtime_error when a Vulkan call fails. +void check(VkResult result, const char* operation); + +// A file next to the executable (SDL base path). +std::string bundled_file_path(const char* name); + +#ifdef __ANDROID__ +void android_haptic(); +#endif + +} // namespace mt_host diff --git a/src/host/input_keymap.cpp b/src/host/input_keymap.cpp new file mode 100644 index 00000000..3d1bb6e7 --- /dev/null +++ b/src/host/input_keymap.cpp @@ -0,0 +1,147 @@ +#include "input_keymap.h" + +namespace mt_host { + + +int sdl_scancode_to_dik(SDL_Scancode sc) { + switch (sc) { + case SDL_SCANCODE_ESCAPE: return 0x01; + case SDL_SCANCODE_1: return 0x02; + case SDL_SCANCODE_2: return 0x03; + case SDL_SCANCODE_3: return 0x04; + case SDL_SCANCODE_4: return 0x05; + case SDL_SCANCODE_5: return 0x06; + case SDL_SCANCODE_6: return 0x07; + case SDL_SCANCODE_7: return 0x08; + case SDL_SCANCODE_8: return 0x09; + case SDL_SCANCODE_9: return 0x0A; + case SDL_SCANCODE_0: return 0x0B; + case SDL_SCANCODE_MINUS: return 0x0C; + case SDL_SCANCODE_EQUALS: return 0x0D; + case SDL_SCANCODE_BACKSPACE: return 0x0E; + case SDL_SCANCODE_TAB: return 0x0F; + case SDL_SCANCODE_Q: return 0x10; + case SDL_SCANCODE_W: return 0x11; + case SDL_SCANCODE_E: return 0x12; + case SDL_SCANCODE_R: return 0x13; + case SDL_SCANCODE_T: return 0x14; + case SDL_SCANCODE_Y: return 0x15; + case SDL_SCANCODE_U: return 0x16; + case SDL_SCANCODE_I: return 0x17; + case SDL_SCANCODE_O: return 0x18; + case SDL_SCANCODE_P: return 0x19; + case SDL_SCANCODE_LEFTBRACKET: return 0x1A; + case SDL_SCANCODE_RIGHTBRACKET: return 0x1B; + case SDL_SCANCODE_RETURN: return 0x1C; + case SDL_SCANCODE_LCTRL: return 0x1D; + case SDL_SCANCODE_A: return 0x1E; + case SDL_SCANCODE_S: return 0x1F; + case SDL_SCANCODE_D: return 0x20; + case SDL_SCANCODE_F: return 0x21; + case SDL_SCANCODE_G: return 0x22; + case SDL_SCANCODE_H: return 0x23; + case SDL_SCANCODE_J: return 0x24; + case SDL_SCANCODE_K: return 0x25; + case SDL_SCANCODE_L: return 0x26; + case SDL_SCANCODE_SEMICOLON: return 0x27; + case SDL_SCANCODE_APOSTROPHE: return 0x28; + case SDL_SCANCODE_GRAVE: return 0x29; + case SDL_SCANCODE_LSHIFT: return 0x2A; + case SDL_SCANCODE_BACKSLASH: return 0x2B; + case SDL_SCANCODE_Z: return 0x2C; + case SDL_SCANCODE_X: return 0x2D; + case SDL_SCANCODE_C: return 0x2E; + case SDL_SCANCODE_V: return 0x2F; + case SDL_SCANCODE_B: return 0x30; + case SDL_SCANCODE_N: return 0x31; + case SDL_SCANCODE_M: return 0x32; + case SDL_SCANCODE_COMMA: return 0x33; + case SDL_SCANCODE_PERIOD: return 0x34; + case SDL_SCANCODE_SLASH: return 0x35; + case SDL_SCANCODE_RSHIFT: return 0x36; + case SDL_SCANCODE_KP_MULTIPLY: return 0x37; + case SDL_SCANCODE_LALT: return 0x38; + case SDL_SCANCODE_SPACE: return 0x39; + case SDL_SCANCODE_CAPSLOCK: return 0x3A; + case SDL_SCANCODE_F1: return 0x3B; + case SDL_SCANCODE_F2: return 0x3C; + case SDL_SCANCODE_F3: return 0x3D; + case SDL_SCANCODE_F4: return 0x3E; + case SDL_SCANCODE_F5: return 0x3F; + case SDL_SCANCODE_F6: return 0x40; + case SDL_SCANCODE_F7: return 0x41; + case SDL_SCANCODE_F8: return 0x42; + case SDL_SCANCODE_F9: return 0x43; + case SDL_SCANCODE_F10: return 0x44; + case SDL_SCANCODE_NUMLOCKCLEAR: return 0x45; + case SDL_SCANCODE_SCROLLLOCK: return 0x46; + case SDL_SCANCODE_KP_7: return 0x47; + case SDL_SCANCODE_KP_8: return 0x48; + case SDL_SCANCODE_KP_9: return 0x49; + case SDL_SCANCODE_KP_MINUS: return 0x4A; + case SDL_SCANCODE_KP_4: return 0x4B; + case SDL_SCANCODE_KP_5: return 0x4C; + case SDL_SCANCODE_KP_6: return 0x4D; + case SDL_SCANCODE_KP_PLUS: return 0x4E; + case SDL_SCANCODE_KP_1: return 0x4F; + case SDL_SCANCODE_KP_2: return 0x50; + case SDL_SCANCODE_KP_3: return 0x51; + case SDL_SCANCODE_KP_0: return 0x52; + case SDL_SCANCODE_KP_PERIOD: return 0x53; + case SDL_SCANCODE_F11: return 0x57; + case SDL_SCANCODE_F12: return 0x58; + case SDL_SCANCODE_KP_ENTER: return 0x9C; + case SDL_SCANCODE_RCTRL: return 0x9D; + case SDL_SCANCODE_KP_DIVIDE: return 0xB5; + case SDL_SCANCODE_RALT: return 0xB8; + case SDL_SCANCODE_HOME: return 0xC7; + case SDL_SCANCODE_UP: return 0xC8; + case SDL_SCANCODE_PAGEUP: return 0xC9; + case SDL_SCANCODE_LEFT: return 0xCB; + case SDL_SCANCODE_RIGHT: return 0xCD; + case SDL_SCANCODE_END: return 0xCF; + case SDL_SCANCODE_DOWN: return 0xD0; + case SDL_SCANCODE_PAGEDOWN: return 0xD1; + case SDL_SCANCODE_INSERT: return 0xD2; + case SDL_SCANCODE_DELETE: return 0xD3; + default: return 0; + } +} + + +int sdl_scancode_to_vk(SDL_Scancode sc) { + switch (sc) { + case SDL_SCANCODE_BACKSPACE: return 0x08; + case SDL_SCANCODE_TAB: return 0x09; + case SDL_SCANCODE_RETURN: + case SDL_SCANCODE_KP_ENTER: return 0x0D; + case SDL_SCANCODE_ESCAPE: return 0x1B; + case SDL_SCANCODE_SPACE: return 0x20; + case SDL_SCANCODE_PAGEUP: return 0x21; + case SDL_SCANCODE_PAGEDOWN: return 0x22; + case SDL_SCANCODE_END: return 0x23; + case SDL_SCANCODE_HOME: return 0x24; + case SDL_SCANCODE_LEFT: return 0x25; + case SDL_SCANCODE_UP: return 0x26; + case SDL_SCANCODE_RIGHT: return 0x27; + case SDL_SCANCODE_DOWN: return 0x28; + case SDL_SCANCODE_INSERT: return 0x2D; + case SDL_SCANCODE_DELETE: return 0x2E; + case SDL_SCANCODE_F1: return 0x70; + case SDL_SCANCODE_F2: return 0x71; + case SDL_SCANCODE_F3: return 0x72; + case SDL_SCANCODE_F4: return 0x73; + case SDL_SCANCODE_F5: return 0x74; + case SDL_SCANCODE_F6: return 0x75; + case SDL_SCANCODE_F7: return 0x76; + case SDL_SCANCODE_F8: return 0x77; + case SDL_SCANCODE_F9: return 0x78; + case SDL_SCANCODE_F10: return 0x79; + case SDL_SCANCODE_F11: return 0x7A; + case SDL_SCANCODE_F12: return 0x7B; + default: return 0; + } +} + + +} // namespace mt_host diff --git a/src/host/input_keymap.h b/src/host/input_keymap.h new file mode 100644 index 00000000..30e214bf --- /dev/null +++ b/src/host/input_keymap.h @@ -0,0 +1,12 @@ +#pragma once + +#include + +namespace mt_host { + +// SDL scancode -> DirectInput DIK_* code (CPythonApplication key events), 0 when unmapped. +int sdl_scancode_to_dik(SDL_Scancode sc); +// SDL scancode -> Win32 VK_* code (WM_KEYDOWN for EditLine / IME), 0 when unmapped. +int sdl_scancode_to_vk(SDL_Scancode sc); + +} // namespace mt_host diff --git a/src/host/live_client.cpp b/src/host/live_client.cpp new file mode 100644 index 00000000..749ac9a8 --- /dev/null +++ b/src/host/live_client.cpp @@ -0,0 +1,514 @@ +#include "live_client.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "android_perf.h" +#include "draw_capture.h" +#include "frame_policy.h" +#include "host_util.h" +#include "native_audio.h" +#include "perf_log.h" +#include "texture_decode.h" +#include "vulkan_window.h" + +#include "platform/MilesLib/AudioCommands.h" +#include "platform/PackBackend.h" +#include "platform/ScriptLib/PythonBoot.h" +#include "platform/UserInterface/ServerClock.h" +#include "platform/EterBase/TraceErrorObserver.h" +#include "platform/EterBase/FrameRateMode.h" +#include "../tests/port/port_login_flow_server.h" + +#include +#ifdef __APPLE__ +#include +#endif + +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace mt_host { + +// --py-exec: a test hook run once the auto-login run reaches the GameWindow (or, with --login-screen, +// once the LoginWindow is up). +std::string g_py_exec_after_login; +// --perf-log / MT_PERF_LOG=1: src/host/perf_log.h. +bool g_perf_log = false; + +bool py_exec(const std::string& code) { + std::string err; + if (!PythonBoot::RunLine(code.c_str(), &err)) { + std::cerr << "PythonBoot::RunLine failed: " << err << '\n'; + return false; + } + return true; +} + +bool py_eval_true(const char* expr) { + std::string res, err; + return PythonBoot::Evaluate(expr, &res, &err) && (res == "True" || res == "1"); +} + +#ifdef __ANDROID__ +bool hide_mobile_taskbar_quickslots() { + return py_exec( + "_mobile_taskbar = _stream.curPhaseWindow.interface.wndTaskBar\n" + "for _mobile_quickslot in _mobile_taskbar.quickslot:\n" + " _mobile_quickslot.Hide()\n" + "_mobile_taskbar.GetChild('QuickSlotBoard').Hide()\n"); +} +#endif + +#if defined(__APPLE__) && TARGET_OS_OSX +bool configure_macos_numeric_quickslots() { + return py_exec( + "_mac_game = _stream.curPhaseWindow\n" + "_mac_game.onPressKeyDict[app.DIK_5] = lambda: _mac_game._GameWindow__PressQuickSlot(4)\n" + "_mac_game.onPressKeyDict[app.DIK_6] = lambda: _mac_game._GameWindow__PressQuickSlot(5)\n" + "_mac_game.onPressKeyDict[app.DIK_7] = lambda: _mac_game._GameWindow__PressQuickSlot(6)\n" + "_mac_game.onPressKeyDict[app.DIK_8] = lambda: _mac_game._GameWindow__PressQuickSlot(7)\n" + "for _mac_key in (app.DIK_F1, app.DIK_F2, app.DIK_F3, app.DIK_F4):\n" + " del _mac_game.onPressKeyDict[_mac_key]\n" + "_mac_taskbar = _mac_game.interface.wndTaskBar\n" + "for _mac_slot_number in range(5, 9):\n" + " _mac_taskbar.GetChild('slot_%d' % _mac_slot_number).LoadImage('d:/ymir work/ui/game/taskbar/%d.sub' % _mac_slot_number)\n"); +} +#endif + +bool parse_live_server_spec(const std::string& spec, std::string& host, int& auth_port, int& game_port) { + const auto p1 = spec.find(':'); + if (p1 == std::string::npos) return false; + const auto p2 = spec.find(':', p1 + 1); + if (p2 == std::string::npos) return false; + host = spec.substr(0, p1); + auth_port = std::stoi(spec.substr(p1 + 1, p2 - p1 - 1)); + game_port = std::stoi(spec.substr(p2 + 1)); + return !host.empty() && auth_port > 0 && game_port > 0; +} + +int run_live_client( + VulkanWindow& renderer, + const std::string& client_dir, + int frames, + int fake_mobs, + bool gpu_skinning, + bool native_terrain, + bool login_screen, + bool selection_screen, + const std::string& live_server_spec, + const std::string& capture_out, + const std::string& screenshot_out) { + if (fake_mobs > 0) { + const std::string mobs_str = std::to_string(fake_mobs); + setenv("MT_FAKE_MOB_COUNT", mobs_str.c_str(), 1); + } +#ifdef __ANDROID__ + SetTraceErrorObserver([](const char* line) { + SDL_LogError(SDL_LOG_CATEGORY_APPLICATION, "40250 SYSERR: %s", line); + }); +#endif + char resolved_client_dir[4096] = {}; + const std::string abs_client_dir = realpath(client_dir.c_str(), resolved_client_dir) ? std::string(resolved_client_dir) : client_dir; + setenv("MT_40250_CLIENT", abs_client_dir.c_str(), 1); +#ifdef __ANDROID__ + SDL_Log("live stage: pack init %s", abs_client_dir.c_str()); +#endif + if (chdir(abs_client_dir.c_str()) != 0 || !mtpack40250::initialize(".")) { + throw std::runtime_error("cannot initialize 40250 pack at: " + abs_client_dir); + } + + const bool host_hardware_cursor = !renderer.touch_controller.is_enabled(); + PythonBoot::SetHostHardwareCursorEnabled(host_hardware_cursor); + if (host_hardware_cursor) renderer.enable_game_hardware_cursor(); + + SetNativeTerrainRenderEnabled(native_terrain); + SetGpuSkinningEnabled(gpu_skinning); + // The shadow map is drawn by the GPU (offscreen pass); MT_CPU_SHADOW=1 / --cpu-shadow keeps the + // CPU rasterizer (RecordingDevice::rasterize_shadow) for comparison. + const char* cpu_shadow = std::getenv("MT_CPU_SHADOW"); + SetGpuRenderTargetsEnabled(!(cpu_shadow && *cpu_shadow == '1')); + NativeAudioEngine audio; +#ifdef __ANDROID__ + SDL_Log("live stage: audio initialized"); +#endif + + std::string server_host = "127.0.0.1"; + int auth_port = 0; + int game_port = 0; + const bool use_external_server = !live_server_spec.empty(); + FakeLoginServer server; + if (use_external_server) { + if (!parse_live_server_spec(live_server_spec, server_host, auth_port, game_port)) + throw std::runtime_error("invalid --live-server spec (expected HOST:AUTH_PORT:GAME_PORT): " + live_server_spec); + } else { + if (!server.Start()) throw std::runtime_error("FakeLoginServer failed to start on loopback"); + auth_port = server.AuthPort(); + game_port = server.GamePort(); + } + + { + // HiDPI text: rasterise glyph pages at the swapchain's pixels per UI pixel (before any font + // exists). Rounded up: a fractional density (Android 1760x800 UI on 2376x1080 = 1.35) then + // scales finer glyphs down instead of stretching 1x ones. MT_FONT_OVERSAMPLE=1 keeps the + // 40250 bilevel 1x glyphs. + const float density = float(renderer.width()) / float(std::max(1u, renderer.logical_width())); + int oversample = std::clamp(int(std::ceil(density - 0.05f)), 1, 4); + if (const char* env = std::getenv("MT_FONT_OVERSAMPLE")) oversample = std::max(1, std::atoi(env)); + UISetFontOversample(oversample); + } + + std::string error; + const char* env_stdlib = std::getenv("MT_PYTHON_STDLIB"); + std::string stdlib = (env_stdlib && *env_stdlib) ? env_stdlib : bundled_file_path("python27.zip"); +#ifdef __ANDROID__ + if (!env_stdlib || !*env_stdlib) { + std::size_t zip_size = 0; + void* zip = SDL_LoadFile("assets://python27.zip", &zip_size); + if (!zip) zip = SDL_LoadFile("python27.zip", &zip_size); + if (!zip) throw std::runtime_error("cannot load bundled python27.zip"); + char* pref = SDL_GetPrefPath("mtgodot", "native-render"); + if (!pref) { SDL_free(zip); throw std::runtime_error("cannot locate app data directory"); } + stdlib = std::string(pref) + "python27.zip"; + SDL_free(pref); + std::ofstream output(stdlib, std::ios::binary | std::ios::trunc); + output.write(static_cast(zip), static_cast(zip_size)); + SDL_free(zip); + if (!output) throw std::runtime_error("cannot extract bundled python27.zip"); + } +#endif + if (!PythonBoot::Start(stdlib.c_str(), &error)) + throw std::runtime_error("PythonBoot::Start: " + error); +#ifdef __ANDROID__ + SDL_Log("live stage: PythonBoot started"); +#endif + + SetPlatformServerTime(123456789); + PythonBoot::SetUISafeInset(renderer.ui_safe_inset()); + PythonBoot::SetUISize(static_cast(renderer.logical_width()), static_cast(renderer.logical_height())); + if (!PythonBoot::RunMainScript("", &error) || !PythonBoot::IsAppLooping()) + throw std::runtime_error("PythonBoot::RunMainScript: " + error); +#ifdef __ANDROID__ + SDL_Log("live stage: main script started"); +#endif + + const std::unordered_map> empty_textures; + auto pump_until = [&](double max_seconds, int sleep_ms, const auto& cond) -> bool { + const auto deadline = std::chrono::steady_clock::now() + std::chrono::duration(max_seconds); + while (std::chrono::steady_clock::now() < deadline && renderer.poll(true) && PythonBoot::IsAppLooping()) { + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + PythonBoot::UIRender(); + audio.pump(); + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); + if (cond()) return true; + if (sleep_ms > 0) + std::this_thread::sleep_for(std::chrono::milliseconds(sleep_ms)); + } + return cond(); + }; + + if (!py_exec("import __main__, app, networkModule, introLogin, introSelect, introLoading, game\n" + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]")) + throw std::runtime_error("failed to locate networkModule.MainStream"); + + if (!pump_until(20.0, 5, [] { return py_eval_true("isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)"); })) + throw std::runtime_error("timed out waiting for LoginWindow"); +#ifdef __ANDROID__ + SDL_Log("live stage: LoginWindow open"); +#endif + + if (login_screen) { + char conn_cmd[512]; + std::snprintf( + conn_cmd, + sizeof(conn_cmd), + "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" + "_w = _stream.curPhaseWindow\n" + "_w._LoginWindow__OpenServerBoard()", + server_host.c_str(), + game_port, + server_host.c_str(), + auth_port); + if (!py_exec(conn_cmd)) + throw std::runtime_error("failed to configure LoginWindow connection info"); + for (int i = 0; i < 15; ++i) { + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + PythonBoot::UIRender(); + } + if (!g_py_exec_after_login.empty() && !py_exec(g_py_exec_after_login)) + throw std::runtime_error("--py-exec failed"); + } else { + char login_cmd[512]; + std::snprintf( + login_cmd, + sizeof(login_cmd), + "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" + "_w = _stream.curPhaseWindow\n" + "_w._LoginWindow__OpenLoginBoard()\n" + "_w.idEditLine.SetText('%s')\n" + "_w.pwdEditLine.SetText('%s')\n" + "_w._LoginWindow__OnClickLoginButton()", + server_host.c_str(), + game_port, + server_host.c_str(), + auth_port, + FakeLoginServer::kLogin, + FakeLoginServer::kPassword); +#ifdef __ANDROID__ + SDL_Log("live stage: submitting fake login"); +#endif + if (!py_exec(login_cmd)) + throw std::runtime_error("failed to submit login credentials"); +#ifdef __ANDROID__ + SDL_Log("live stage: fake login submitted"); +#endif + + if (!pump_until(20.0, 5, [&] { + return (use_external_server || server.Has("game1:login2")) && + py_eval_true("isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)"); + })) + throw std::runtime_error("timed out waiting for SelectCharacterWindow"); +#ifdef __ANDROID__ + SDL_Log("live stage: SelectCharacterWindow open"); +#endif + + if (!selection_screen) { + if (!py_exec("_stream.curPhaseWindow.SelectSlot(0)\n" + "_stream.curPhaseWindow.StartGame()")) + throw std::runtime_error("failed to start game from SelectCharacterWindow"); + + if (!use_external_server) { + if (!pump_until(30.0, 5, [&] { return server.Has("game2:client_version"); })) + throw std::runtime_error("timed out waiting for LoadingWindow (client_version)"); + } + + if (!pump_until(60.0, 0, [&] { + return (use_external_server || server.Has("game2:burst_pong")) && + py_eval_true("isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()"); + })) + throw std::runtime_error("timed out waiting for GameWindow (server error: " + server.Error() + ")"); + +#ifdef __ANDROID__ + if (!hide_mobile_taskbar_quickslots()) + throw std::runtime_error("failed to hide the desktop quick-slot bar on Android"); +#endif +#if defined(__APPLE__) && TARGET_OS_OSX + if (!configure_macos_numeric_quickslots()) + throw std::runtime_error("failed to configure numeric quick slots on macOS"); +#endif + if (!g_py_exec_after_login.empty() && !py_exec(g_py_exec_after_login)) + throw std::runtime_error("--py-exec failed"); + + const int warmup_frames = std::max(10, fake_mobs / 2 + 10); + for (int i = 0; i < warmup_frames; ++i) { + if (!renderer.poll(true) || !PythonBoot::IsAppLooping()) break; + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + audio.pump(); + } + } + } + + // Render one warmup frame to upload initial scene & UI textures/geometries before timed benchmark. + { + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); + } + + if (!capture_out.empty()) { + std::unordered_map> cap_textures; + auto add_tex = [&](const std::string& name) { + if (name.empty() || cap_textures.count(name)) return; + if (name.rfind("mem:", 0) == 0) { + UIMemoryTexture mem_tex; + if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { + auto mtra = native_draw_capture::encode_raw_argb_as_mtra( + static_cast(mem_tex.width), + static_cast(mem_tex.height), + mem_tex.argb.data()); + if (!mtra.empty()) cap_textures.emplace(name, std::move(mtra)); + } + return; + } + std::vector bytes; + if (read_live_pack_texture(name, bytes)) + cap_textures.emplace(name, std::move(bytes)); + }; + for (const auto& d : Render3DDraws()) { + add_tex(d.texture0); + add_tex(d.texture1); + } + for (const auto& c : UIRenderCommands()) { + if (c.kind == UIRenderCommand::Image) { + add_tex(c.text); + add_tex(c.mask); + } + } + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + std::vector cap_ui = UIRenderCommands(); + if (renderer.touch_controller.is_enabled()) { + renderer.touch_controller.update_screen_size(int(ui_w), int(ui_h)); + renderer.touch_controller.append_ui_commands(cap_ui); + } + native_draw_capture::write(capture_out, Render3DDraws(), cap_textures, ui_w, ui_h, cap_ui); + } + + if (!screenshot_out.empty() && frames > 0) { + renderer.request_screenshot(screenshot_out, renderer.frame_number() + static_cast(frames)); + } + + renderer.reset_timings(); + native_perf::PerfLog perf_log; + if (g_perf_log) { + perf_log.open("device=" + renderer.device_name() + " present_mode=" + renderer.present_mode_name() + + " msaa=" + std::to_string(renderer.msaa_samples()) + " size=" + std::to_string(renderer.width()) + + "x" + std::to_string(renderer.height()) + " gpu_skinning=" + std::to_string(gpu_skinning) + + " terrain=" + std::to_string(native_terrain)); + } + if (g_refresh_rate > 0) { + const bool ok = android_perf::request_frame_rate(renderer.sdl_window(), float(g_refresh_rate)); + SDL_Log("frame rate request %d Hz: %s", g_refresh_rate, ok ? "accepted" : "unavailable"); + } + perf_log.set_fps_cap(g_fps_cap); + perf_log.set_refresh_rate_source([] { return android_perf::display_refresh_rate(); }); + perf_log.set_render_rate_source([] { return android_perf::render_rate(); }); + perf_log.set_battery_source([] { return android_perf::battery_celsius(); }); + perf_log.set_power_source([] { + const auto p = android_perf::battery_power(); + return native_perf::PowerSample{p.current_ma, p.voltage_mv, p.power_mw, p.plugged}; + }); + // ADPF: each frame's CPU work (everything but the vsync/fence waits and the cap's sleep) against the + // frame budget. Opened once the script thread exists, i.e. here. + using steady = std::chrono::steady_clock; + // The target is 80% of the frame period: the governor settles the clocks so the reported work just + // meets the target, so a target equal to the period leaves every other frame late (cap 60 on the + // test phone: 54 fps against the period, 60 against 80% of it). + const auto budget_for = [](int fps) { return std::chrono::nanoseconds(800'000'000LL / std::max(1, fps)); }; + FrameRatePolicy policy; + policy.load(); + const auto frame_budget = budget_for( + g_fps_cap > 0 ? g_fps_cap : (g_refresh_rate > 0 ? g_refresh_rate : 60)); + android_perf::PerformanceHint hint; + { + std::vector tids{android_perf::current_thread_id()}; + if (const int script = PythonBoot::ScriptThreadId()) tids.push_back(script); + const bool ok = hint.open(tids, frame_budget.count()); + SDL_Log("ADPF performance hint (%zu threads, %.2f ms): %s", tids.size(), frame_budget.count() / 1e6, + ok ? "on" : "unavailable"); + } + const auto start = steady::now(); + double update_ms = 0.0; + int completed = 0; + std::vector frame_ms; + if (frames > 0) frame_ms.reserve(static_cast(frames)); + auto cap_period = g_fps_cap > 0 ? std::chrono::nanoseconds(1'000'000'000LL / g_fps_cap) + : std::chrono::nanoseconds(0); + auto next_frame_at = steady::now(); + FrameRatePolicy::Decision decision; + auto display_checked_at = steady::time_point{}; + // The loop sleeps to the policy's rate unless the display already runs at it (vsync then paces the + // loop, and a second clock beating against it would drop frames). + const auto apply_policy = [&] { + if (policy.pinned()) return; + const auto now = steady::now(); + const auto next = policy.decide(renderer.input_idle_seconds(), PythonBoot::AutoHuntIsEnabled()); + const bool changed = !(next == decision); + if (changed) { + if (next.display_hz != decision.display_hz) { + android_perf::set_display_refresh_rate(float(next.display_hz)); + android_perf::request_frame_rate(renderer.sdl_window(), float(next.display_hz)); + } + hint.set_target(budget_for(next.fps).count()); + // Efficiency cores and lower clocks, except where the setting asks for the highest rate. + hint.prefer_power_efficiency(next.mode != MtFrameRate::MODE_HIGH || next.idle); + perf_log.set_fps_cap(next.fps); + perf_log.set_frame_policy(next.mode, next.idle); + SDL_Log("frame rate: mode %d -> %d fps, display %d Hz%s", next.mode, next.fps, next.display_hz, + next.idle ? " (idle)" : ""); + decision = next; + } + if (changed || now - display_checked_at > std::chrono::seconds(1)) { + display_checked_at = now; + const double hz = android_perf::display_refresh_rate(); + const bool vsync_paced = hz > 0 && std::fabs(hz - decision.fps) < 2.0; + const auto period = vsync_paced ? std::chrono::nanoseconds(0) + : std::chrono::nanoseconds(1'000'000'000LL / decision.fps); + if (period != cap_period) { + cap_period = period; + next_frame_at = now; + } + } + }; + while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) { + apply_policy(); + double pace_ms = 0.0; + if (cap_period.count()) { + // Fixed cadence: a late frame starts the next one at once, a frame more than one period + // late re-anchors instead of bursting to catch up. + const auto now = steady::now(); + if (next_frame_at > now) { + std::this_thread::sleep_until(next_frame_at); + pace_ms = std::chrono::duration(steady::now() - now).count(); + } else if (now - next_frame_at > cap_period) { + next_frame_at = now; + } + next_frame_at += cap_period; + } + const auto frame_start = steady::now(); + const auto t0 = frame_start; + PythonBoot::UIUpdate(); + const auto t_app = std::chrono::steady_clock::now(); + renderer.sync_game_cursor(); + PythonBoot::UIRender(); + audio.pump(); + const auto t1 = std::chrono::steady_clock::now(); + update_ms += std::chrono::duration(t1 - t0).count(); + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); + const auto t2 = std::chrono::steady_clock::now(); + const double elapsed_ms = std::chrono::duration(t2 - frame_start).count(); + PythonBoot::SetNativeRenderFrameTime(static_cast(elapsed_ms)); + hint.report(std::int64_t((elapsed_ms - renderer.last_sync_ms()) * 1e6)); + if (perf_log.enabled()) { + native_perf::FrameInput in; + in.frame_ms = elapsed_ms; + in.pace_ms = pace_ms; + in.app_ms = std::chrono::duration(t_app - t0).count(); + in.host_ms = std::chrono::duration(t1 - t_app).count(); + in.render_ms = std::chrono::duration(t2 - t1).count(); + in.renderer = renderer.perf_totals(); + perf_log.frame(in, [] { return PythonBoot::CurrentMapName(); }); + } + frame_ms.push_back(elapsed_ms); + ++completed; + } + renderer.finish_gpu_timings(); +#ifdef __ANDROID__ + SDL_Log("live stage: render loop complete (%d frames)", completed); +#endif + const auto wall_ms = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); + print_summary(renderer, completed, wall_ms, update_ms, std::move(frame_ms)); + + PythonBoot::Stop(); +#ifdef __ANDROID__ + SDL_Log("live stage: PythonBoot stopped"); +#endif + if (!use_external_server) + server.Stop(); + return (frames == 0 || completed == frames) ? 0 : 2; +} + +} // namespace mt_host +#endif diff --git a/src/host/live_client.h b/src/host/live_client.h new file mode 100644 index 00000000..0a9c5260 --- /dev/null +++ b/src/host/live_client.h @@ -0,0 +1,31 @@ +#pragma once + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include + +namespace mt_host { + +class VulkanWindow; + +// --py-exec: a test hook run once the auto-login run reaches the GameWindow (or, with --login-screen, +// once the LoginWindow is up). +extern std::string g_py_exec_after_login; +// --perf-log / MT_PERF_LOG=1: src/host/perf_log.h. +extern bool g_perf_log; + +// Boots the 40250 client (PythonBoot) from client_dir and runs it inside the renderer until it quits. +int run_live_client( + VulkanWindow& renderer, + const std::string& client_dir, + int frames, + int fake_mobs, + bool gpu_skinning, + bool native_terrain, + bool login_screen, + bool selection_screen, + const std::string& live_server_spec, + const std::string& capture_out, + const std::string& screenshot_out); + +} // namespace mt_host +#endif diff --git a/src/host/main.cpp b/src/host/main.cpp index 2497a7e6..7a8f0b86 100644 --- a/src/host/main.cpp +++ b/src/host/main.cpp @@ -1,3697 +1,28 @@ -#include "RenderCommands3D.h" -#include "UIRenderCommands.h" #include "draw_capture.h" -#include "dxt.h" -#include "touch_controller.h" +#include "frame_policy.h" +#include "host_util.h" +#include "live_client.h" #include "perf_log.h" -#include "android_perf.h" - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT -#include "platform/MilesLib/AudioCommands.h" -#include "platform/PackBackend.h" -#include "platform/ScriptLib/PythonBoot.h" -#include "platform/UserInterface/ServerClock.h" -#include "platform/EterBase/TraceErrorObserver.h" -#include "platform/EterBase/FrameRateMode.h" -#include "../tests/port/port_login_flow_server.h" -#include -#endif +#include "vulkan_window.h" #include #include #include -#include -#ifdef __ANDROID__ -#include -#endif -#include -#include "stb_image.h" - -#ifdef __APPLE__ -#include -#include -#endif #include -#include #include -#include -#include #include -#include #include -#include -#include #include -#include -#include #include #include -#include -#include #include +using namespace mt_host; + namespace { -void check(VkResult result, const char* operation) { - if (result != VK_SUCCESS) throw std::runtime_error(std::string(operation) + ": VkResult " + std::to_string(result)); -} - -struct Vertex { - float position[3]; - float normal[3]; - float uv[2]; - float color[4]; - std::uint8_t joints[4]; - float weights[4]; - float mask_uv[2]; - float rhw = 1.0f; - float vertex_fog = 1.0f; -}; - -struct PushConstants { - std::array mvp{}; - // x: bone palette base + 1 (0 = not skinned), y: D3DFVF_XYZRHW, z: UI mask batch, w: unused. - std::array params{}; -}; -static_assert(sizeof(PushConstants) <= 128, "PushConstants must fit within Vulkan's 128-byte minimum guarantee"); - -// std430 payload selected with a dynamic storage-buffer offset for every draw: the D3D8 -// fixed-function state in effect for the draw (texture stages, texture coordinate processing, -// lighting, material, fog, alpha test). A zero flags.x marks a 2D UI batch. -struct alignas(16) FixedFunctionState { - std::array, 2> stage_color{}; // op, arg1, arg2, D3DTSS_TEXCOORDINDEX - std::array, 2> stage_alpha{}; // op, arg1, arg2, D3DTSS_TEXTURETRANSFORMFLAGS - std::array fog_color{}; - std::array fog_params{}; // start, end, density, range enabled - std::array texture_factor{}; - std::array world_view{}; - std::array flags{}; // fixed function, texture0, texture1, fog mode - std::array normal_matrix{}; // inverse transpose of world * view - std::array lighting_flags{}; // LIGHTING, COLORVERTEX, vertex has diffuse, NORMALIZENORMALS - std::array material_sources{}; // DIFFUSE, AMBIENT, EMISSIVE source, LOCALVIEWER - std::array alpha_test{}; // ALPHATESTENABLE, ALPHAFUNC, ALPHAREF, unused - std::array material_diffuse{}; - std::array material_ambient{}; - std::array material_emissive{}; - std::array global_ambient{}; - std::array, 2> texture_matrix{}; - // Lights in camera space. position.w = D3DLIGHTTYPE (0 = disabled). - std::array, 8> light_position_type{}; - std::array, 8> light_direction_range{}; - std::array, 8> light_diffuse{}; - std::array, 8> light_ambient{}; - std::array, 8> light_attenuation{}; // a0, a1, a2, falloff - std::array, 8> light_spot{}; // cos(theta / 2), cos(phi / 2) -}; -static_assert(sizeof(FixedFunctionState) == 1264); - -constexpr std::array kIdentityMatrix = { - 1.0f, 0.0f, 0.0f, 0.0f, - 0.0f, 1.0f, 0.0f, 0.0f, - 0.0f, 0.0f, 1.0f, 0.0f, - 0.0f, 0.0f, 0.0f, 1.0f}; - -std::array multiply(const float* a, const float* b) { - std::array result{}; - for (int row = 0; row < 4; ++row) - for (int col = 0; col < 4; ++col) - for (int k = 0; k < 4; ++k) - result[row * 4 + col] += a[row * 4 + k] * b[k * 4 + col]; - return result; -} - -std::array draw_mvp(const Render3DDraw& draw) { - const auto world_view = multiply(draw.world, draw.view); - return multiply(world_view.data(), draw.proj); -} - -std::array unpack_argb(std::uint32_t argb) { - return { - float((argb >> 16) & 255) / 255.0f, - float((argb >> 8) & 255) / 255.0f, - float(argb & 255) / 255.0f, - float((argb >> 24) & 255) / 255.0f}; -} - -// Inverse transpose of the upper 3x3 of a row-vector matrix, the D3D8 normal transform. -std::array normal_matrix(const std::array& m) { - const float a = m[0], b = m[1], c = m[2], d = m[4], e = m[5], f = m[6], g = m[8], h = m[9], i = m[10]; - const float A = e * i - f * h, B = -(d * i - f * g), C = d * h - e * g; - const float D = -(b * i - c * h), E = a * i - c * g, F = -(a * h - b * g); - const float G = b * f - c * e, H = -(a * f - c * d), I = a * e - b * d; - const float det = a * A + b * B + c * C; - std::array out = kIdentityMatrix; - if (det == 0.0f) return out; - const float s = 1.0f / det; - // inverse = adjugate / det; its transpose is the cofactor matrix / det. - out[0] = A * s; out[1] = B * s; out[2] = C * s; - out[4] = D * s; out[5] = E * s; out[6] = F * s; - out[8] = G * s; out[9] = H * s; out[10] = I * s; - return out; -} - -PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded) { - PushConstants constants{}; - constants.mvp = draw_mvp(draw); - constants.params = {skin_offset_encoded, draw.pretransformed ? 1.0f : 0.0f, 0.0f, 0.0f}; - return constants; -} - -FixedFunctionState make_fixed_function_state(const Render3DDraw& draw) { - FixedFunctionState state{}; - for (int stage = 0; stage < 2; ++stage) { - state.stage_color[stage] = {draw.color_op[stage], draw.color_arg1[stage], draw.color_arg2[stage], - draw.texcoord_index[stage]}; - state.stage_alpha[stage] = {draw.alpha_op[stage], draw.alpha_arg1[stage], draw.alpha_arg2[stage], - draw.texture_transform_flags[stage]}; - std::copy(draw.texture_matrix[stage], draw.texture_matrix[stage] + 16, state.texture_matrix[stage].begin()); - } - state.texture_factor = unpack_argb(draw.texture_factor); - state.world_view = multiply(draw.world, draw.view); - state.normal_matrix = normal_matrix(state.world_view); - state.fog_color = unpack_argb(draw.fog_color); - state.fog_params = {draw.fog_start, draw.fog_end, draw.fog_density, - draw.fog_range_enable ? 1.0f : 0.0f}; - const std::uint32_t fog_mode = draw.fog_enable - ? (draw.fog_table_mode ? draw.fog_table_mode : (draw.pretransformed ? 4u : draw.fog_vertex_mode)) : 0; - state.flags = {1u, !draw.texture0.empty() ? 1u : 0u, !draw.texture1.empty() ? 1u : 0u, fog_mode}; - state.lighting_flags = {draw.lighting, draw.color_vertex, draw.diffuse.empty() ? 0u : 1u, - draw.normalize_normals}; - state.material_sources = {draw.diffuse_material_source, draw.ambient_material_source, - draw.emissive_material_source, draw.local_viewer}; - state.alpha_test = {draw.alpha_test, draw.alpha_func, draw.alpha_ref, 0}; - std::copy(draw.material_diffuse, draw.material_diffuse + 4, state.material_diffuse.begin()); - std::copy(draw.material_ambient, draw.material_ambient + 4, state.material_ambient.begin()); - std::copy(draw.material_emissive, draw.material_emissive + 4, state.material_emissive.begin()); - state.global_ambient = unpack_argb(draw.ambient); - const float* v = draw.view; - for (int i = 0; i < 8; ++i) { - const auto& light = draw.lights[i]; - const float* p = light.position; - const float* d = light.direction; - std::array position{}, direction{}; - for (int c = 0; c < 3; ++c) { - position[c] = p[0] * v[c] + p[1] * v[4 + c] + p[2] * v[8 + c] + v[12 + c]; - direction[c] = d[0] * v[c] + d[1] * v[4 + c] + d[2] * v[8 + c]; - } - const float length = std::sqrt(direction[0] * direction[0] + direction[1] * direction[1] + - direction[2] * direction[2]); - if (length > 0.0f) - for (float& c : direction) c /= length; - state.light_position_type[i] = {position[0], position[1], position[2], float(light.type)}; - state.light_direction_range[i] = {direction[0], direction[1], direction[2], light.range}; - std::copy(light.diffuse, light.diffuse + 4, state.light_diffuse[i].begin()); - std::copy(light.ambient, light.ambient + 4, state.light_ambient[i].begin()); - state.light_attenuation[i] = {light.attenuation[0], light.attenuation[1], light.attenuation[2], light.falloff}; - state.light_spot[i] = {std::cos(light.theta * 0.5f), std::cos(light.phi * 0.5f), 0, 0}; - } - return state; -} - -std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size) { - const auto* data = static_cast(bytes); - std::size_t i = 0; - for (; i + sizeof(std::uint64_t) <= size; i += sizeof(std::uint64_t)) { - std::uint64_t word = 0; - std::memcpy(&word, data + i, sizeof(word)); - hash = (hash ^ word) * 1099511628211ull; - } - for (; i < size; ++i) hash = (hash ^ data[i]) * 1099511628211ull; - hash = (hash ^ size) * 1099511628211ull; - return hash; -} - -// Immediate-mode draws have no source-buffer key. For them, verify content before reusing a slot. -std::uint64_t geometry_hash(const Render3DDraw& draw) { - std::uint64_t hash = 14695981039346656037ull; - const std::uint32_t flags = (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u); - hash = hash_bytes(hash, &flags, sizeof(flags)); - hash = hash_bytes(hash, draw.positions.data(), draw.positions.size() * sizeof(float)); - hash = hash_bytes(hash, draw.rhw.data(), draw.rhw.size() * sizeof(float)); - hash = hash_bytes(hash, draw.vertex_fog.data(), draw.vertex_fog.size() * sizeof(float)); - hash = hash_bytes(hash, draw.normals.data(), draw.normals.size() * sizeof(float)); - hash = hash_bytes(hash, draw.uv0.data(), draw.uv0.size() * sizeof(float)); - hash = hash_bytes(hash, draw.uv1.data(), draw.uv1.size() * sizeof(float)); - hash = hash_bytes(hash, draw.diffuse.data(), draw.diffuse.size() * sizeof(std::uint32_t)); - return hash_bytes(hash, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); -} - -mtimage::Image decode_tga(const std::uint8_t* data, std::size_t size) { - mtimage::Image out; - if (size < 18) return out; - const std::uint8_t id_len = data[0]; - const std::uint8_t cmap_type = data[1]; - const std::uint8_t img_type = data[2]; - const std::uint32_t w = std::uint32_t(data[12]) | (std::uint32_t(data[13]) << 8); - const std::uint32_t h = std::uint32_t(data[14]) | (std::uint32_t(data[15]) << 8); - const std::uint8_t bpp = data[16]; - const std::uint8_t desc = data[17]; - if (cmap_type != 0 || !w || !h || w > 4096 || h > 4096) return out; - if (img_type != 2 && img_type != 3 && img_type != 10) return out; - const std::size_t bytes_per_pixel = bpp / 8; - if (bytes_per_pixel != 1 && bytes_per_pixel != 3 && bytes_per_pixel != 4) return out; - std::size_t offset = 18 + std::size_t(id_len); - if (offset > size) return out; - - const std::size_t pixel_count = std::size_t(w) * std::size_t(h); - std::vector temp(pixel_count * 4); - auto write_pixel = [&](std::size_t idx, const std::uint8_t* src) { - std::uint8_t* dst = &temp[idx * 4]; - if (bytes_per_pixel == 1) { - dst[0] = dst[1] = dst[2] = src[0]; - dst[3] = 255; - } else if (bytes_per_pixel == 3) { - dst[0] = src[2]; - dst[1] = src[1]; - dst[2] = src[0]; - dst[3] = 255; - } else { - dst[0] = src[2]; - dst[1] = src[1]; - dst[2] = src[0]; - dst[3] = src[3]; - } - }; - - if (img_type == 2 || img_type == 3) { - if (offset + pixel_count * bytes_per_pixel > size) return out; - for (std::size_t i = 0; i < pixel_count; ++i) - write_pixel(i, data + offset + i * bytes_per_pixel); - } else if (img_type == 10) { - std::size_t i = 0; - while (i < pixel_count && offset < size) { - const std::uint8_t header = data[offset++]; - const std::size_t run = (header & 0x7fu) + 1u; - if (header & 0x80u) { - if (offset + bytes_per_pixel > size) return out; - for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) - write_pixel(i, data + offset); - offset += bytes_per_pixel; - } else { - if (offset + run * bytes_per_pixel > size) return out; - for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) { - write_pixel(i, data + offset); - offset += bytes_per_pixel; - } - } - } - if (i != pixel_count) return out; - } - - out.w = static_cast(w); - out.h = static_cast(h); - const bool top_origin = (desc & 0x20u) != 0; - if (top_origin) { - out.rgba = std::move(temp); - } else { - out.rgba.resize(pixel_count * 4); - const std::size_t row_bytes = std::size_t(w) * 4; - for (std::uint32_t y = 0; y < h; ++y) - std::memcpy(&out.rgba[std::size_t(y) * row_bytes], &temp[std::size_t(h - 1 - y) * row_bytes], row_bytes); - } - return out; -} - -mtimage::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size) { - if (!data || size < 12) return {}; - if (data[0] == 'M' && data[1] == 'T' && data[2] == 'R' && data[3] == 'A') { - std::uint32_t w = 0, h = 0; - std::memcpy(&w, data + 4, 4); - std::memcpy(&h, data + 8, 4); - const std::size_t bytes = std::size_t(w) * std::size_t(h) * 4; - if (w > 0 && h > 0 && w <= 4096 && h <= 4096 && size == 12 + bytes) { - mtimage::Image out; - out.w = static_cast(w); - out.h = static_cast(h); - out.rgba.assign(data + 12, data + 12 + bytes); - return out; - } - return {}; - } - if (data[0] == 'D' && data[1] == 'D' && data[2] == 'S' && data[3] == ' ') - return mtimage::load_dds(data, size); - auto tga = decode_tga(data, size); - if (tga.ok()) return tga; - if (size > static_cast(INT_MAX)) return {}; - int w = 0, h = 0, channels = 0; - stbi_uc* pixels = stbi_load_from_memory(data, static_cast(size), &w, &h, &channels, 4); - if (pixels && w > 0 && h > 0 && w <= 4096 && h <= 4096) { - mtimage::Image out; - out.w = static_cast(w); - out.h = static_cast(h); - out.rgba.assign(pixels, pixels + std::size_t(w) * std::size_t(h) * 4); - stbi_image_free(pixels); - return out; - } - stbi_image_free(pixels); - return {}; -} - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT -bool read_live_pack_texture(const std::string& vpath, std::vector& bytes) { - if (vpath.empty() || !mtpack40250::ready()) - return false; - std::string norm = vpath; - for (char& ch : norm) - if (ch == '\\') ch = '/'; - std::string stripped = norm; - if (stripped.size() >= 2 && stripped[1] == ':') - stripped = stripped.substr(2); - while (!stripped.empty() && stripped.front() == '/') - stripped.erase(stripped.begin()); - auto lower = [](std::string s) { - for (char& ch : s) - if (ch >= 'A' && ch <= 'Z') - ch = static_cast(ch - 'A' + 'a'); - return s; - }; - for (const std::string& candidate : { - norm, stripped, "d:/" + stripped, - lower(norm), lower(stripped), lower("d:/" + stripped)}) { - if (mtpack40250::read(candidate, bytes) && !bytes.empty()) - return true; - } - return false; -} - -SDL_Cursor* load_game_cursor(int shape) { - static constexpr std::array images = { - "cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub", - "cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub", - "cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub", - "cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"}; - if (shape < 0 || shape >= static_cast(images.size())) return nullptr; - const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape]; - std::vector sub_bytes; - if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr; - std::unordered_map tokens; - std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end())); - std::string line; - while (std::getline(input, line)) { - std::istringstream fields(line); - std::string key, value; - if (!(fields >> key >> value)) continue; - if (!value.empty() && value.front() == '"') { - value.erase(0, 1); - if (!value.empty() && value.back() == '"') value.pop_back(); - } - std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); }); - tokens[key] = value; - } - if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr; - const std::string image_path = tokens["version"] == "2.0" - ? std::string("d:/ymir work/ui/cursor/") + tokens["image"] - : std::string("d:/ymir work/ui/") + tokens["image"]; - std::vector image_bytes; - if (!read_live_pack_texture(image_path, image_bytes)) return nullptr; - const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size()); - if (!image.ok()) return nullptr; - const int left = std::atoi(tokens["left"].c_str()); - const int top = std::atoi(tokens["top"].c_str()); - const int right = std::atoi(tokens["right"].c_str()); - const int bottom = std::atoi(tokens["bottom"].c_str()); - if (left < 0 || top < 0 || right <= left || bottom <= top || - right > image.w || bottom > image.h) return nullptr; - const int width = right - left, height = bottom - top; - std::vector pixels(std::size_t(width) * height * 4); - for (int y = 0; y < height; ++y) - std::memcpy(pixels.data() + std::size_t(y) * width * 4, - image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4, - std::size_t(width) * 4); - SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32, - pixels.data(), width * 4); - if (!surface) return nullptr; - const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0; - const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0; - SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y); - SDL_DestroySurface(surface); - return cursor; -} - -class NativeAudioEngine { - struct MemoryAudioBuffer { - const std::uint8_t* data = nullptr; - std::size_t size = 0; - }; - struct Voice { - const std::vector* samples = nullptr; - std::size_t frame_cursor = 0; - float volume = 1.0f; - float target_volume = 1.0f; - float fade_step_per_frame = 0.0f; - bool stop_after_fade = false; - bool loop = false; - bool is_3d = false; - int handle = 0; - std::string filename; - }; - -#ifdef __APPLE__ - static OSStatus mem_audio_read_proc( - void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) { - const auto* mem = static_cast(inClientData); - if (inPosition < 0 || static_cast(inPosition) >= mem->size) { - *actualCount = 0; - return noErr; - } - const std::size_t avail = mem->size - static_cast(inPosition); - const std::size_t to_read = std::min(requestCount, avail); - std::memcpy(buffer, mem->data + inPosition, to_read); - *actualCount = static_cast(to_read); - return noErr; - } - - static SInt64 mem_audio_get_size_proc(void* inClientData) { - return static_cast(static_cast(inClientData)->size); - } -#endif - -public: - NativeAudioEngine() { - if (SDL_InitSubSystem(SDL_INIT_AUDIO)) { - SDL_AudioSpec spec{}; - spec.format = SDL_AUDIO_F32; - spec.channels = 2; - spec.freq = 44100; - stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_PLAYBACK, &spec, nullptr, nullptr); - if (stream_) - SDL_ResumeAudioStreamDevice(stream_); - } - // Always DrainAudioCommands() from cold boot so stale queued commands don't accumulate. - (void)DrainAudioCommands(); - } - - ~NativeAudioEngine() { - if (stream_) - SDL_DestroyAudioStream(stream_); - SDL_QuitSubSystem(SDL_INIT_AUDIO); - } - - void pump() { - const auto commands = DrainAudioCommands(); - for (const auto& cmd : commands) { - ++commands_drained_; - switch (cmd.type) { - case AudioCommand::PlaySound2D: { - const auto* clip = get_clip(cmd.filename); - if (clip && !clip->empty()) { - Voice v{}; - v.samples = clip; - v.volume = sound_volume_; - v.target_volume = sound_volume_; - v.filename = cmd.filename; - voices_.push_back(std::move(v)); - ++played_2d_; - } - break; - } - case AudioCommand::PlaySound3D: { - const auto* clip = get_clip(cmd.filename); - if (clip && !clip->empty()) { - Voice v{}; - v.samples = clip; - v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); - v.target_volume = v.volume; - v.loop = cmd.play_count != 1; - v.is_3d = true; - v.handle = cmd.id; - v.filename = cmd.filename; - voices_.push_back(std::move(v)); - ++played_3d_; - } - break; - } - case AudioCommand::StopSound3D: - voices_.erase( - std::remove_if(voices_.begin(), voices_.end(), [&](const Voice& v) { - return v.is_3d && v.handle == cmd.id; - }), - voices_.end()); - break; - case AudioCommand::StopAllSound3D: - voices_.erase( - std::remove_if(voices_.begin(), voices_.end(), [](const Voice& v) { return v.is_3d; }), - voices_.end()); - break; - case AudioCommand::SetSoundVolume3D: - for (auto& v : voices_) { - if (v.is_3d && v.handle == cmd.id) { - v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); - v.target_volume = v.volume; - } - } - break; - case AudioCommand::PlayMusic: - case AudioCommand::FadeInMusic: { - const auto* clip = get_clip(cmd.filename); - if (clip && !clip->empty()) { - bgm_.samples = clip; - bgm_.frame_cursor = 0; - bgm_.loop = true; - bgm_.filename = cmd.filename; - bgm_.stop_after_fade = false; - const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); - bgm_.target_volume = target; - if (cmd.type == AudioCommand::FadeInMusic) { - bgm_.volume = 0.0f; - bgm_.fade_step_per_frame = target / (44100.0f * 1.5f); - } else { - bgm_.volume = target; - bgm_.fade_step_per_frame = 0.0f; - } - ++played_music_; - } - break; - } - case AudioCommand::FadeOutMusic: - case AudioCommand::FadeOutAllMusic: - if (bgm_.samples) { - bgm_.target_volume = 0.0f; - bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f); - bgm_.stop_after_fade = true; - } - break; - case AudioCommand::FadeLimitOutMusic: - if (bgm_.samples) { - bgm_.target_volume = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); - bgm_.fade_step_per_frame = (bgm_.target_volume - bgm_.volume) / (44100.0f * 1.0f); - bgm_.stop_after_fade = false; - } - break; - case AudioCommand::SetMusicVolume: - music_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); - if (bgm_.samples && !bgm_.stop_after_fade) { - bgm_.volume = music_volume_; - bgm_.target_volume = music_volume_; - } - break; - case AudioCommand::SetSoundVolume: - sound_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); - break; - default: - break; - } - } - - if (!stream_) return; - const int queued_bytes = SDL_GetAudioStreamAvailable(stream_); - if (queued_bytes < 0) return; - const std::size_t queued_frames = static_cast(queued_bytes) / (2 * sizeof(float)); - constexpr std::size_t kTargetQueuedFrames = 4410; // ~100 ms stereo @ 44.1 kHz - if (queued_frames >= kTargetQueuedFrames) return; - const std::size_t frames_to_mix = kTargetQueuedFrames - queued_frames; - mix_buffer_.assign(frames_to_mix * 2, 0.0f); - - auto mix_voice = [&](Voice& v) -> bool { - if (!v.samples || v.samples->empty()) return false; - const std::size_t total_frames = v.samples->size() / 2; - if (total_frames == 0) return false; - const float* src = v.samples->data(); - for (std::size_t f = 0; f < frames_to_mix; ++f) { - if (v.frame_cursor >= total_frames) { - if (v.loop) v.frame_cursor = 0; - else return false; - } - if (v.fade_step_per_frame != 0.0f) { - v.volume += v.fade_step_per_frame; - if ((v.fade_step_per_frame > 0.0f && v.volume >= v.target_volume) || - (v.fade_step_per_frame < 0.0f && v.volume <= v.target_volume)) { - v.volume = v.target_volume; - v.fade_step_per_frame = 0.0f; - if (v.stop_after_fade && v.volume <= 0.0001f) { - v.samples = nullptr; - return false; - } - } - } - mix_buffer_[f * 2 + 0] += src[v.frame_cursor * 2 + 0] * v.volume; - mix_buffer_[f * 2 + 1] += src[v.frame_cursor * 2 + 1] * v.volume; - ++v.frame_cursor; - } - return v.loop || v.frame_cursor < total_frames; - }; - - if (bgm_.samples) { - if (!mix_voice(bgm_)) bgm_.samples = nullptr; - } - for (auto it = voices_.begin(); it != voices_.end();) { - if (!mix_voice(*it)) it = voices_.erase(it); - else ++it; - } - for (float& sample : mix_buffer_) - sample = std::clamp(sample, -1.0f, 1.0f); - SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast(mix_buffer_.size() * sizeof(float))); - } - -private: - const std::vector* get_clip(const std::string& vpath) { - if (vpath.empty()) return nullptr; - auto [it, inserted] = clips_.try_emplace(vpath); - if (!inserted) return &it->second; -#ifdef __APPLE__ - std::vector bytes; - if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16) - return &it->second; - MemoryAudioBuffer mem{bytes.data(), bytes.size()}; - AudioFileID audio_file = nullptr; - if (AudioFileOpenWithCallbacks( - &mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr || - !audio_file) { - return &it->second; - } - ExtAudioFileRef ext_file = nullptr; - if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) { - AudioFileClose(audio_file); - return &it->second; - } - AudioStreamBasicDescription client_format{}; - client_format.mSampleRate = 44100.0; - client_format.mFormatID = kAudioFormatLinearPCM; - client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked; - client_format.mBytesPerPacket = 8; - client_format.mFramesPerPacket = 1; - client_format.mBytesPerFrame = 8; - client_format.mChannelsPerFrame = 2; - client_format.mBitsPerChannel = 32; - if (ExtAudioFileSetProperty( - ext_file, - kExtAudioFileProperty_ClientDataFormat, - sizeof(client_format), - &client_format) == noErr) { - std::vector chunk(4096 * 2); - while (true) { - UInt32 frame_count = 4096; - AudioBufferList buf_list{}; - buf_list.mNumberBuffers = 1; - buf_list.mBuffers[0].mNumberChannels = 2; - buf_list.mBuffers[0].mDataByteSize = static_cast(chunk.size() * sizeof(float)); - buf_list.mBuffers[0].mData = chunk.data(); - if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0) - break; - it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2); - if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes - break; - } - } - ExtAudioFileDispose(ext_file); - AudioFileClose(audio_file); -#endif - return &it->second; - } - - SDL_AudioStream* stream_ = nullptr; - std::unordered_map> clips_; - std::vector voices_; - Voice bgm_{}; - std::vector mix_buffer_; - float sound_volume_ = 0.8f; - float music_volume_ = 0.5f; - std::size_t commands_drained_ = 0; - std::size_t played_2d_ = 0; - std::size_t played_3d_ = 0; - std::size_t played_music_ = 0; -}; -#endif - -std::string bundled_file_path(const char* name) { - const char* base = SDL_GetBasePath(); - return std::string(base ? base : "./") + name; -} - -#ifdef __ANDROID__ -// MainActivity.performHaptic(): the long-press vibration of the touch gestures. -void android_haptic() { - auto* env = static_cast(SDL_GetAndroidJNIEnv()); - auto activity = static_cast(SDL_GetAndroidActivity()); - if (!env || !activity) return; - jclass cls = env->GetObjectClass(activity); - if (jmethodID method = env->GetMethodID(cls, "performHaptic", "()V")) - env->CallVoidMethod(activity, method); - if (env->ExceptionCheck()) env->ExceptionClear(); - env->DeleteLocalRef(cls); - env->DeleteLocalRef(activity); -} -#endif - -std::vector read_spirv(const char* name) { - std::size_t size = 0; - void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size); -#ifdef __ANDROID__ - if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size); - if (!bytes) bytes = SDL_LoadFile(name, &size); -#endif - if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name); - if (size == 0 || size % 4) { - SDL_free(bytes); - throw std::runtime_error(std::string("invalid SPIR-V size: ") + name); - } - std::vector words(size / 4); - std::memcpy(words.data(), bytes, size); - SDL_free(bytes); - return words; -} - -class VulkanWindow { - struct FrameSlot; - struct GeometryId { - std::uint64_t key = 0, signature = 0; - bool operator==(const GeometryId&) const = default; - }; - struct GeometryIdHash { - std::size_t operator()(GeometryId id) const { - return std::size_t(id.key ^ (id.signature + 0x9e3779b97f4a7c15ull + (id.key << 6) + (id.key >> 2))); - } - }; - struct Geometry { - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - VkDeviceSize index_offset = 0; - std::uint64_t last_used_frame = 0; - std::uint32_t vertex_count = 0, index_count = 0; - }; - struct GpuTexture { - VkImage image = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - VkImageView view = VK_NULL_HANDLE; - VkDescriptorSet descriptor = VK_NULL_HANDLE; - VkDescriptorSet ui_descriptor = VK_NULL_HANDLE; - std::uint64_t last_used_frame = 0; - std::uint64_t last_decode_attempt_frame = 0; - }; - struct PairedDescriptor { - VkDescriptorSet set = VK_NULL_HANDLE; - std::uint64_t last_used_frame = 0; - }; - // An offscreen D3D render-target texture (the 40250 character shadow map, "rt::x"): - // drawn by the frame's offscreen pass ahead of the back-buffer pass and sampled by that pass. - // Between frames the colour image stays SHADER_READ_ONLY_OPTIMAL and the depth image - // DEPTH_STENCIL_ATTACHMENT_OPTIMAL; the render pass keeps (LOAD/STORE) both, like a D3D surface. - struct RenderTarget { - GpuTexture texture; // colour image, view and descriptors used for sampling - VkImage depth_image = VK_NULL_HANDLE; - VkDeviceMemory depth_memory = VK_NULL_HANDLE; - VkImageView depth_view = VK_NULL_HANDLE; - VkFramebuffer framebuffer = VK_NULL_HANDLE; - std::uint32_t width = 0, height = 0; - bool initialized = false; // cleared to white / depth 1 by the first frame that uses it - }; - struct PreparedDraw { - Geometry* geometry = nullptr; - VkPipeline pipeline = VK_NULL_HANDLE; - VkDescriptorSet descriptor = VK_NULL_HANDLE; - PushConstants constants{}; - FixedFunctionState fixed_state{}; - std::uint32_t state_offset = 0; - VkViewport viewport{}; - // Offscreen render target of the draw, or null for the back buffer. - RenderTarget* target = nullptr; - // IDirect3DDevice8::Clear recorded in draw order; geometry is null for these. - std::uint32_t clear_flags = 0; - VkClearValue clear_color{}; - VkClearValue clear_depth{}; - VkRect2D clear_rect{}; - }; - struct UiBatch { - std::uint32_t first_index = 0; - std::uint32_t index_count = 0; - VkPipeline pipeline = VK_NULL_HANDLE; - VkDescriptorSet descriptor = VK_NULL_HANDLE; - PushConstants constants{}; - bool behind_3d = false; - std::uint32_t state_offset = 0; - }; - - static constexpr std::size_t kMaxBonesPerFrame = 65536; - static constexpr VkDeviceSize kBoneBufferBytes = kMaxBonesPerFrame * 16 * sizeof(float); - static constexpr std::size_t kMaxUiVerticesPerFrame = 65536; - static constexpr std::size_t kMaxUiIndicesPerFrame = 98304; - static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex); - static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t); - static constexpr VkDeviceSize kStagingRingBytes = 4ull * 1024 * 1024; - static constexpr VkDeviceSize kStagingRingMaxBytes = 64ull * 1024 * 1024; - -public: - struct Timings { - double sync_ms = 0; - double fence_ms = 0; // inside sync_ms: waiting for the frame slot's previous GPU work - double prepare_ms = 0; - double steady_prepare_ms = 0; - double submit_ms = 0; - double present_ms = 0; - double gpu_ms = 0; - std::uint64_t gpu_samples = 0; - // Inside prepare_ms: creating geometry buffers and the blocking texture uploads. - double geometry_upload_ms = 0; - double texture_upload_ms = 0; - }; - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - void enable_game_hardware_cursor() { - hardware_cursor_enabled_ = true; - for (int shape = 0; shape < static_cast(game_cursors_.size()); ++shape) { - game_cursors_[shape] = load_game_cursor(shape); - } - fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT); - SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_); - SDL_ShowCursor(); - os_cursor_hidden_ = false; - } - - void sync_game_cursor() { - if (!hardware_cursor_enabled_) return; - const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() && - !PythonBoot::IsSoftwareCursorVisible(); - if (visible == os_cursor_hidden_) { - if (visible) SDL_ShowCursor(); - else SDL_HideCursor(); - os_cursor_hidden_ = !visible; - } - if (!visible) return; - const int shape = PythonBoot::CursorShape(); - if (shape == current_cursor_shape_) return; - SDL_Cursor* cursor = shape >= 0 && shape < static_cast(game_cursors_.size()) - ? game_cursors_[shape] : nullptr; - if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_); - current_cursor_shape_ = shape; - } -#endif - - explicit VulkanWindow(bool vsync = true, int init_width = 960, int init_height = 640) : vsync_(vsync) { -#ifdef __APPLE__ - if (!std::getenv("VK_ICD_FILENAMES")) { - for (const char* icd : { - "/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json", - "/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json"}) { - if (std::ifstream(icd).good()) { - setenv("VK_ICD_FILENAMES", icd, 0); - break; - } - } - } -#endif - // Android's back key reaches the game as SDL_SCANCODE_AC_BACK (-> ESC) instead of finishing the activity. - SDL_SetHint(SDL_HINT_ANDROID_TRAP_BACK_BUTTON, "1"); - if (!SDL_Init(SDL_INIT_VIDEO)) throw std::runtime_error(SDL_GetError()); - SDL_SetHint(SDL_HINT_ORIENTATIONS, "LandscapeLeft LandscapeRight"); - window_ = SDL_CreateWindow( - "Metin2 Native Vulkan Client", - std::max(init_width, 320), - std::max(init_height, 240), - SDL_WINDOW_VULKAN | SDL_WINDOW_RESIZABLE | SDL_WINDOW_HIGH_PIXEL_DENSITY); - if (!window_) throw std::runtime_error(SDL_GetError()); - SDL_StartTextInput(window_); - Uint32 extension_count = 0; - const char* const* sdl_extensions = SDL_Vulkan_GetInstanceExtensions(&extension_count); - if (!sdl_extensions) throw std::runtime_error(SDL_GetError()); - std::vector extensions(sdl_extensions, sdl_extensions + extension_count); - std::uint32_t available_count = 0; - check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, nullptr), "vkEnumerateInstanceExtensionProperties count"); - std::vector available(available_count); - check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, available.data()), "vkEnumerateInstanceExtensionProperties"); - const bool portability = std::any_of(available.begin(), available.end(), [](const auto& extension) { - return std::strcmp(extension.extensionName, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; - }); - if (portability && std::find_if(extensions.begin(), extensions.end(), [](const char* name) { - return std::strcmp(name, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; - }) == extensions.end()) - extensions.push_back(VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME); - VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO}; - app.pApplicationName = "Metin2 native renderer"; - app.apiVersion = VK_API_VERSION_1_1; - VkInstanceCreateInfo instance_info{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO}; - instance_info.flags = portability ? VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR : 0; - instance_info.pApplicationInfo = &app; - instance_info.enabledExtensionCount = static_cast(extensions.size()); - instance_info.ppEnabledExtensionNames = extensions.data(); - check(vkCreateInstance(&instance_info, nullptr, &instance_), "vkCreateInstance"); - if (!SDL_Vulkan_CreateSurface(window_, instance_, nullptr, &surface_)) throw std::runtime_error(SDL_GetError()); - select_device(); - VkCommandPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; - pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; - pool_info.queueFamilyIndex = queue_family_; - check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool"); - VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO}; - query_info.queryType = VK_QUERY_TYPE_TIMESTAMP; - query_info.queryCount = 2 * kMaxFramesInFlight; - check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool"); - VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; - VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; - fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; - for (std::uint32_t i = 0; i < kMaxFramesInFlight; ++i) { - auto& slot = frames_[i]; - VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; - alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1; - check(vkAllocateCommandBuffers(device_, &alloc, &slot.command), "vkAllocateCommandBuffers"); - check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &slot.acquire), "vkCreateSemaphore acquire"); - check(vkCreateFence(device_, &fence_info, nullptr, &slot.fence), "vkCreateFence"); - slot.query_base = 2 * i; - } - create_swapchain(); - create_descriptors_and_buffers(); - create_render_pass_and_layout(); - touch_controller.update_screen_size(int(width()), int(height())); - } - - ~VulkanWindow() { - if (hardware_cursor_enabled_) { - SDL_SetCursor(nullptr); - for (SDL_Cursor* cursor : game_cursors_) if (cursor) SDL_DestroyCursor(cursor); - if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_); - } - if (device_) vkDeviceWaitIdle(device_); - for (auto& slot : frames_) { - release_retired_buffers(slot); - if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); - if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); - if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); - if (slot.bone_mapped) vkUnmapMemory(device_, slot.bone_memory); - if (slot.bone_buffer) vkDestroyBuffer(device_, slot.bone_buffer, nullptr); - if (slot.bone_memory) vkFreeMemory(device_, slot.bone_memory, nullptr); - if (slot.state_mapped) vkUnmapMemory(device_, slot.state_memory); - if (slot.state_buffer) vkDestroyBuffer(device_, slot.state_buffer, nullptr); - if (slot.state_memory) vkFreeMemory(device_, slot.state_memory, nullptr); - if (slot.ui_mapped) vkUnmapMemory(device_, slot.ui_memory); - if (slot.ui_buffer) vkDestroyBuffer(device_, slot.ui_buffer, nullptr); - if (slot.ui_memory) vkFreeMemory(device_, slot.ui_memory, nullptr); - if (slot.fence) vkDestroyFence(device_, slot.fence, nullptr); - if (slot.acquire) vkDestroySemaphore(device_, slot.acquire, nullptr); - } - if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr); - for (VkSemaphore semaphore : rendered_) vkDestroySemaphore(device_, semaphore, nullptr); - for (auto& entry : geometries_) release_geometry(entry.second); - for (auto& entry : textures_) release_texture(entry.second); - for (auto& entry : render_targets_) release_render_target(entry.second); - if (offscreen_pass_) vkDestroyRenderPass(device_, offscreen_pass_, nullptr); - release_texture(fallback_texture_); - for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); - if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr); - if (fragment_module_) vkDestroyShaderModule(device_, fragment_module_, nullptr); - if (layout_) vkDestroyPipelineLayout(device_, layout_, nullptr); - if (descriptor_pool_) vkDestroyDescriptorPool(device_, descriptor_pool_, nullptr); - if (descriptor_layout_) vkDestroyDescriptorSetLayout(device_, descriptor_layout_, nullptr); - if (bone_descriptor_layout_) vkDestroyDescriptorSetLayout(device_, bone_descriptor_layout_, nullptr); - if (sampler_) vkDestroySampler(device_, sampler_, nullptr); - if (clamp_sampler_) vkDestroySampler(device_, clamp_sampler_, nullptr); - if (ui_sampler_) vkDestroySampler(device_, ui_sampler_, nullptr); - for (const auto& entry : state_samplers_) vkDestroySampler(device_, entry.second, nullptr); - for (auto framebuffer : framebuffers_) vkDestroyFramebuffer(device_, framebuffer, nullptr); - if (pass_) vkDestroyRenderPass(device_, pass_, nullptr); - if (depth_view_) vkDestroyImageView(device_, depth_view_, nullptr); - if (depth_image_) vkDestroyImage(device_, depth_image_, nullptr); - if (depth_memory_) vkFreeMemory(device_, depth_memory_, nullptr); - destroy_msaa_color(); - for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); - if (swapchain_) vkDestroySwapchainKHR(device_, swapchain_, nullptr); - if (pool_) vkDestroyCommandPool(device_, pool_, nullptr); - if (device_) vkDestroyDevice(device_, nullptr); - if (surface_) vkDestroySurfaceKHR(instance_, surface_, nullptr); - if (instance_) vkDestroyInstance(instance_, nullptr); - if (window_) { - SDL_StopTextInput(window_); - SDL_DestroyWindow(window_); - } - SDL_Quit(); - } - - // 32-bit bottom-up BMP from a B8G8R8A8 / R8G8B8A8 swapchain readback. - void write_bmp(const std::string& path, const std::uint8_t* pixels) const { - const std::uint32_t w = extent_.width, h = extent_.height; - const bool rgba = swapchain_format_ == VK_FORMAT_R8G8B8A8_UNORM || swapchain_format_ == VK_FORMAT_R8G8B8A8_SRGB; - std::vector file(54 + std::size_t(w) * h * 4); - auto put32 = [&](std::size_t at, std::uint32_t value) { std::memcpy(&file[at], &value, 4); }; - file[0] = 'B'; file[1] = 'M'; - put32(2, std::uint32_t(file.size())); put32(10, 54); put32(14, 40); - put32(18, w); put32(22, h); put32(26, 1u | (32u << 16)); put32(34, w * h * 4); - for (std::uint32_t y = 0; y < h; ++y) - for (std::uint32_t x = 0; x < w; ++x) { - const std::uint8_t* src = pixels + (std::size_t(y) * w + x) * 4; - std::uint8_t* dst = &file[54 + (std::size_t(h - 1 - y) * w + x) * 4]; - dst[0] = rgba ? src[2] : src[0]; dst[1] = src[1]; dst[2] = rgba ? src[0] : src[2]; dst[3] = 255; - } - std::ofstream out(path, std::ios::binary); - out.write(reinterpret_cast(file.data()), std::streamsize(file.size())); - } - - // Seconds since the last touch, mouse or key event; 0 while a finger or mouse button is held (a held - // joystick or attack button sends no events). The frame-rate policy's idle test. - double input_idle_seconds() const { - if (fingers_down_ > 0 || buttons_down_ > 0) return 0.0; - return std::chrono::duration(std::chrono::steady_clock::now() - last_input_).count(); - } - - void request_screenshot(std::string path, std::uint64_t frame) { - screenshot_path_ = std::move(path); - screenshot_frame_ = frame; - } - std::uint64_t frame_number() const { return frame_number_; } - - int msaa_samples() const { return int(samples_); } - std::uint32_t width() const { return extent_.width; } - std::uint32_t height() const { return extent_.height; } - std::uint32_t logical_width() const { - int w = 0, h = 0; - if (window_) SDL_GetWindowSize(window_, &w, &h); -#ifdef __ANDROID__ - // Keep the 40250 UI at the same height as the macOS 1280x800 window, - // while retaining the device aspect ratio on wider phone displays. - if (w > 0 && h > 0) return std::max(1, int(std::lround(double(w) * 800.0 / h))); -#endif - return w > 0 ? static_cast(w) : extent_.width; - } - std::uint32_t logical_height() const { - int w = 0, h = 0; - if (window_) SDL_GetWindowSize(window_, &w, &h); -#ifdef __ANDROID__ - if (w > 0 && h > 0) return 800; -#endif - return h > 0 ? static_cast(h) : extent_.height; - } - - VkViewport full_viewport() const { - return {0, 0, float(extent_.width), float(extent_.height), 0, 1}; - } - - // Window pixels kept clear at the left and right screen edges (rounded corners, cutouts); - // MainActivity measures them and passes --safe-inset-px. - void set_safe_inset_px(int px) { safe_inset_px_ = std::max(0, px); } - // Test builds: frames per second in the top-left corner (--show-fps). - void set_show_fps(bool show) { show_fps_ = show; } - SDL_Window* sdl_window() const { return window_; } - // The last render()'s blocking time: slot fence + swapchain acquire. - double last_sync_ms() const { return last_sync_ms_; } - // The same inset in logical UI pixels: the 40250 window layers are shifted and narrowed by it. - int ui_safe_inset() const { - int pw = 0, ph = 0; - if (safe_inset_px_ <= 0 || !window_ || !SDL_GetWindowSizeInPixels(window_, &pw, &ph) || ph <= 0) return 0; - return int(std::lround(double(safe_inset_px_) * logical_height() / ph)); - } - - // D3DVIEWPORT8 (in the game's logical screen pixels) scaled to the swapchain extent. - VkViewport draw_viewport(const Render3DDraw& draw) const { - if (draw.viewport[2] <= 0 || draw.viewport[3] <= 0) return full_viewport(); - const float sx = float(extent_.width) / float(logical_width()); - const float sy = float(extent_.height) / float(logical_height()); - return {(draw.viewport[0] + draw.ui_offset_x) * sx, draw.viewport[1] * sy, draw.viewport[2] * sx, draw.viewport[3] * sy, - draw.viewport_z[0], draw.viewport_z[1]}; - } - - TouchController touch_controller; - - static int sdl_scancode_to_dik(SDL_Scancode sc) { - switch (sc) { - case SDL_SCANCODE_ESCAPE: return 0x01; - case SDL_SCANCODE_1: return 0x02; - case SDL_SCANCODE_2: return 0x03; - case SDL_SCANCODE_3: return 0x04; - case SDL_SCANCODE_4: return 0x05; - case SDL_SCANCODE_5: return 0x06; - case SDL_SCANCODE_6: return 0x07; - case SDL_SCANCODE_7: return 0x08; - case SDL_SCANCODE_8: return 0x09; - case SDL_SCANCODE_9: return 0x0A; - case SDL_SCANCODE_0: return 0x0B; - case SDL_SCANCODE_MINUS: return 0x0C; - case SDL_SCANCODE_EQUALS: return 0x0D; - case SDL_SCANCODE_BACKSPACE: return 0x0E; - case SDL_SCANCODE_TAB: return 0x0F; - case SDL_SCANCODE_Q: return 0x10; - case SDL_SCANCODE_W: return 0x11; - case SDL_SCANCODE_E: return 0x12; - case SDL_SCANCODE_R: return 0x13; - case SDL_SCANCODE_T: return 0x14; - case SDL_SCANCODE_Y: return 0x15; - case SDL_SCANCODE_U: return 0x16; - case SDL_SCANCODE_I: return 0x17; - case SDL_SCANCODE_O: return 0x18; - case SDL_SCANCODE_P: return 0x19; - case SDL_SCANCODE_LEFTBRACKET: return 0x1A; - case SDL_SCANCODE_RIGHTBRACKET: return 0x1B; - case SDL_SCANCODE_RETURN: return 0x1C; - case SDL_SCANCODE_LCTRL: return 0x1D; - case SDL_SCANCODE_A: return 0x1E; - case SDL_SCANCODE_S: return 0x1F; - case SDL_SCANCODE_D: return 0x20; - case SDL_SCANCODE_F: return 0x21; - case SDL_SCANCODE_G: return 0x22; - case SDL_SCANCODE_H: return 0x23; - case SDL_SCANCODE_J: return 0x24; - case SDL_SCANCODE_K: return 0x25; - case SDL_SCANCODE_L: return 0x26; - case SDL_SCANCODE_SEMICOLON: return 0x27; - case SDL_SCANCODE_APOSTROPHE: return 0x28; - case SDL_SCANCODE_GRAVE: return 0x29; - case SDL_SCANCODE_LSHIFT: return 0x2A; - case SDL_SCANCODE_BACKSLASH: return 0x2B; - case SDL_SCANCODE_Z: return 0x2C; - case SDL_SCANCODE_X: return 0x2D; - case SDL_SCANCODE_C: return 0x2E; - case SDL_SCANCODE_V: return 0x2F; - case SDL_SCANCODE_B: return 0x30; - case SDL_SCANCODE_N: return 0x31; - case SDL_SCANCODE_M: return 0x32; - case SDL_SCANCODE_COMMA: return 0x33; - case SDL_SCANCODE_PERIOD: return 0x34; - case SDL_SCANCODE_SLASH: return 0x35; - case SDL_SCANCODE_RSHIFT: return 0x36; - case SDL_SCANCODE_KP_MULTIPLY: return 0x37; - case SDL_SCANCODE_LALT: return 0x38; - case SDL_SCANCODE_SPACE: return 0x39; - case SDL_SCANCODE_CAPSLOCK: return 0x3A; - case SDL_SCANCODE_F1: return 0x3B; - case SDL_SCANCODE_F2: return 0x3C; - case SDL_SCANCODE_F3: return 0x3D; - case SDL_SCANCODE_F4: return 0x3E; - case SDL_SCANCODE_F5: return 0x3F; - case SDL_SCANCODE_F6: return 0x40; - case SDL_SCANCODE_F7: return 0x41; - case SDL_SCANCODE_F8: return 0x42; - case SDL_SCANCODE_F9: return 0x43; - case SDL_SCANCODE_F10: return 0x44; - case SDL_SCANCODE_NUMLOCKCLEAR: return 0x45; - case SDL_SCANCODE_SCROLLLOCK: return 0x46; - case SDL_SCANCODE_KP_7: return 0x47; - case SDL_SCANCODE_KP_8: return 0x48; - case SDL_SCANCODE_KP_9: return 0x49; - case SDL_SCANCODE_KP_MINUS: return 0x4A; - case SDL_SCANCODE_KP_4: return 0x4B; - case SDL_SCANCODE_KP_5: return 0x4C; - case SDL_SCANCODE_KP_6: return 0x4D; - case SDL_SCANCODE_KP_PLUS: return 0x4E; - case SDL_SCANCODE_KP_1: return 0x4F; - case SDL_SCANCODE_KP_2: return 0x50; - case SDL_SCANCODE_KP_3: return 0x51; - case SDL_SCANCODE_KP_0: return 0x52; - case SDL_SCANCODE_KP_PERIOD: return 0x53; - case SDL_SCANCODE_F11: return 0x57; - case SDL_SCANCODE_F12: return 0x58; - case SDL_SCANCODE_KP_ENTER: return 0x9C; - case SDL_SCANCODE_RCTRL: return 0x9D; - case SDL_SCANCODE_KP_DIVIDE: return 0xB5; - case SDL_SCANCODE_RALT: return 0xB8; - case SDL_SCANCODE_HOME: return 0xC7; - case SDL_SCANCODE_UP: return 0xC8; - case SDL_SCANCODE_PAGEUP: return 0xC9; - case SDL_SCANCODE_LEFT: return 0xCB; - case SDL_SCANCODE_RIGHT: return 0xCD; - case SDL_SCANCODE_END: return 0xCF; - case SDL_SCANCODE_DOWN: return 0xD0; - case SDL_SCANCODE_PAGEDOWN: return 0xD1; - case SDL_SCANCODE_INSERT: return 0xD2; - case SDL_SCANCODE_DELETE: return 0xD3; - default: return 0; - } - } - - static int sdl_scancode_to_vk(SDL_Scancode sc) { - switch (sc) { - case SDL_SCANCODE_BACKSPACE: return 0x08; - case SDL_SCANCODE_TAB: return 0x09; - case SDL_SCANCODE_RETURN: - case SDL_SCANCODE_KP_ENTER: return 0x0D; - case SDL_SCANCODE_ESCAPE: return 0x1B; - case SDL_SCANCODE_SPACE: return 0x20; - case SDL_SCANCODE_PAGEUP: return 0x21; - case SDL_SCANCODE_PAGEDOWN: return 0x22; - case SDL_SCANCODE_END: return 0x23; - case SDL_SCANCODE_HOME: return 0x24; - case SDL_SCANCODE_LEFT: return 0x25; - case SDL_SCANCODE_UP: return 0x26; - case SDL_SCANCODE_RIGHT: return 0x27; - case SDL_SCANCODE_DOWN: return 0x28; - case SDL_SCANCODE_INSERT: return 0x2D; - case SDL_SCANCODE_DELETE: return 0x2E; - case SDL_SCANCODE_F1: return 0x70; - case SDL_SCANCODE_F2: return 0x71; - case SDL_SCANCODE_F3: return 0x72; - case SDL_SCANCODE_F4: return 0x73; - case SDL_SCANCODE_F5: return 0x74; - case SDL_SCANCODE_F6: return 0x75; - case SDL_SCANCODE_F7: return 0x76; - case SDL_SCANCODE_F8: return 0x77; - case SDL_SCANCODE_F9: return 0x78; - case SDL_SCANCODE_F10: return 0x79; - case SDL_SCANCODE_F11: return 0x7A; - case SDL_SCANCODE_F12: return 0x7B; - default: return 0; - } - } - - // Touch hosts: SDL text input (and with it the system keyboard) is on only while the player - // types into a tapped EditLine; desktop keeps the always-on text input from window creation. - void sync_screen_keyboard() { - const bool want = touch_controller.wants_screen_keyboard(); - const unsigned serial = touch_controller.screen_keyboard_serial(); - if (want && (!SDL_TextInputActive(window_) || serial != screen_keyboard_serial_)) { - SDL_PropertiesID props = SDL_CreateProperties(); - SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_TYPE_NUMBER, SDL_TEXTINPUT_TYPE_TEXT); - SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_CAPITALIZATION_NUMBER, SDL_CAPITALIZE_NONE); - SDL_SetBooleanProperty(props, SDL_PROP_TEXTINPUT_AUTOCORRECT_BOOLEAN, false); - SDL_StartTextInputWithProperties(window_, props); - SDL_DestroyProperties(props); - } else if (!want && SDL_TextInputActive(window_)) { - SDL_StopTextInput(window_); - } - screen_keyboard_serial_ = serial; - } - - bool poll(bool forward_to_live_client = false) { -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - auto map_mouse = [&](float wx, float wy, int& ox, int& oy) { - int win_w = 0, win_h = 0; - SDL_GetWindowSize(window_, &win_w, &win_h); - unsigned ui_w = extent_.width ? extent_.width : 960; - unsigned ui_h = extent_.height ? extent_.height : 640; - UIRenderGetSize(&ui_w, &ui_h); - if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) { - ox = static_cast(std::lround(wx * float(ui_w) / float(win_w))); - oy = static_cast(std::lround(wy * float(ui_h) / float(win_h))); - } else { - ox = static_cast(std::lround(wx)); - oy = static_cast(std::lround(wy)); - } - }; - std::optional> pending_mouse_move; - auto flush_mouse_move = [&] { - if (!pending_mouse_move) return; - const auto [mx, my] = *pending_mouse_move; - pending_mouse_move.reset(); - last_mx_ = mx; - last_my_ = my; - if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return; - PythonBoot::UIMouseMove(mx, my); - }; -#endif - SDL_Event event; - while (SDL_PollEvent(&event)) { - note_input(event); -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - // Touches arrive as SDL_EVENT_FINGER_*; the touch-synthesized mouse copy - // would register as a second finger (101) and turn every drag into a pinch. - if (touch_controller.is_enabled() && - ((event.type == SDL_EVENT_MOUSE_MOTION && event.motion.which == SDL_TOUCH_MOUSEID) || - ((event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) && - event.button.which == SDL_TOUCH_MOUSEID))) - continue; - if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) { - int mx = 0, my = 0; - map_mouse(event.motion.x, event.motion.y, mx, my); - pending_mouse_move = std::pair{mx, my}; - continue; - } - // Preserve event order at button/key/focus boundaries while collapsing - // high-frequency motion into the most recent position. - flush_mouse_move(); -#endif - if (event.type == SDL_EVENT_QUIT) return false; - if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) { - int ew = 0, eh = 0; - SDL_GetWindowSize(window_, &ew, &eh); - touch_controller.update_screen_size(int(logical_width()), int(logical_height()), ui_safe_inset()); -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (forward_to_live_client && ew > 0 && eh > 0) { - PythonBoot::SetUISafeInset(ui_safe_inset()); - PythonBoot::SetUISize(int(logical_width()), int(logical_height())); - } -#endif - recreate_swapchain(); - continue; - } - if (event.type == SDL_EVENT_FINGER_DOWN) { - touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); - continue; - } else if (event.type == SDL_EVENT_FINGER_MOTION) { - touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); - continue; - } else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) { - touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); - continue; - } -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (forward_to_live_client) { - if (!hardware_cursor_enabled_ && - (event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) { - std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n"); - SDL_HideCursor(); - os_cursor_hidden_ = true; - } else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) { - SDL_CaptureMouse(false); - PythonBoot::UIMouseButton(2, false, last_mx_, last_my_); - PythonBoot::UIMouseButton(3, false, last_mx_, last_my_); - PythonBoot::UIMouseButton(1, false, last_mx_, last_my_); - } - if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) { - const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN; - int btn = 0; - if (event.button.button == SDL_BUTTON_LEFT) btn = 1; - else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2; - else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3; - if (btn) { - int mx = 0, my = 0; - map_mouse(event.button.x, event.button.y, mx, my); - last_mx_ = mx; - last_my_ = my; - if (touch_controller.is_enabled()) { - if (touch_controller.on_mouse_button(btn, pressed, mx, my)) { - continue; - } - } - SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0); - PythonBoot::UIMouseButton(btn, pressed, mx, my); - } - } else if (event.type == SDL_EVENT_MOUSE_WHEEL) { - PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f)); - } else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) { - const bool pressed = event.type == SDL_EVENT_KEY_DOWN; - // Android back = ESC: closes the top window, or opens the system menu (game.py). - if (event.key.scancode == SDL_SCANCODE_AC_BACK) event.key.scancode = SDL_SCANCODE_ESCAPE; - if (pressed) { - if (const int vk = sdl_scancode_to_vk(event.key.scancode)) - PythonBoot::UIIMEKeyDown(vk); - if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8); - else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9); - else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER) - PythonBoot::UIChar(13); - else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27); - } - if (!event.key.repeat) { - if (const int dik = sdl_scancode_to_dik(event.key.scancode)) - PythonBoot::UIKey(dik, pressed); - } - } else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) { - const auto* s = reinterpret_cast(event.text.text); - while (*s) { - unsigned cp = 0; - if (*s < 0x80u) { - cp = *s++; - } else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) { - cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu); - s += 2; - } else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) { - cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu); - s += 3; - } else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) { - cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu); - s += 4; - } else { - ++s; - continue; - } - if (cp >= 32u && cp != 127u) - PythonBoot::UIChar(cp); - } - } - } -#else - (void)forward_to_live_client; -#endif - } -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - flush_mouse_move(); -#endif - touch_controller.update(); - if (touch_controller.is_enabled()) - sync_screen_keyboard(); -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) { - SDL_HideCursor(); - os_cursor_hidden_ = true; - } -#endif - return true; - } - - void reset_timings() { - finish_gpu_timings(); - timings_ = {}; - timed_frames_ = 0; - upload_count_ = 0; - uploaded_bytes_ = 0; - texture_upload_count_ = 0; - texture_uploaded_bytes_ = 0; - } - - void finish_gpu_timings() { - for (auto& slot : frames_) { - if (!slot.has_pending_query) continue; - check(vkWaitForFences(device_, 1, &slot.fence, VK_TRUE, UINT64_MAX), "vkWaitForFences finish"); - collect_pending_gpu_timestamp(slot); - } - } - - // 1 = the old serial loop (CPU waits for the previous frame's GPU work), 2 = CPU/GPU overlap. - void set_frames_in_flight(std::uint32_t count) { - frames_in_flight_ = std::clamp(count, 1, kMaxFramesInFlight); - } - std::uint32_t frames_in_flight() const { return frames_in_flight_; } - - void render( - const std::vector& draws, - const std::unordered_map>& capture_textures, - std::uint32_t ui_width = 960, - std::uint32_t ui_height = 640, - const std::vector& ui_commands = {}) { - if (extent_.width == 0 || extent_.height == 0) return; - const auto start = std::chrono::steady_clock::now(); - f_ = &frames_[frame_slot_ % frames_in_flight_]; - check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences"); - collect_pending_gpu_timestamp(*f_); - reset_staging(*f_); - const auto fenced = std::chrono::steady_clock::now(); - prune_textures(); - std::uint32_t image_index = 0; - const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, f_->acquire, VK_NULL_HANDLE, &image_index); - if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { - recreate_swapchain(); - return; - } - if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR"); - const auto synchronized = std::chrono::steady_clock::now(); - ++frame_number_; - ++timed_frames_; - while (rendered_.size() < swapchain_images_.size()) { - VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; - VkSemaphore semaphore = VK_NULL_HANDLE; - check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &semaphore), "vkCreateSemaphore rendered"); - rendered_.push_back(semaphore); - } - // Texture uploads of this frame are recorded ahead of the render pass (upload_texture). - check(vkResetCommandBuffer(f_->command, 0), "vkResetCommandBuffer"); - VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; - begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; - check(vkBeginCommandBuffer(f_->command, &begin), "vkBeginCommandBuffer"); - vkCmdResetQueryPool(f_->command, query_pool_, f_->query_base, 2); - vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, f_->query_base); - f_->recording = true; - - std::vector prepared_draws; - prepared_draws.reserve(draws.size()); - // Draws into offscreen render targets (the shadow map), in recorded order, drawn before the - // back-buffer pass that samples them. - std::vector offscreen_draws; - std::unordered_set active_keys; - std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; - std::uint32_t bone_cursor = 0; - std::size_t required_bones = 0; - for (const auto& draw : draws) { - if (!draw.bone_matrices.empty()) { - if (draw.bone_matrices.size() % 16) - throw std::runtime_error("invalid bone palette length"); - required_bones += draw.bone_matrices.size() / 16; - } - } - ensure_bone_capacity(required_bones); - - for (const auto& draw : draws) { - RenderTarget* target = nullptr; - if (draw.render_target) { - target = get_render_target( - Render3DRenderTargetName(draw.render_target, draw.target_width, draw.target_height)); - if (!target) continue; - } - if (draw.clear_flags && target) { - PreparedDraw clear{}; - clear.target = target; - clear.clear_flags = draw.clear_flags; - const auto color = unpack_argb(draw.clear_color); - clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; - clear.clear_depth.depthStencil = {draw.clear_z, 0}; - clear.clear_rect = {{0, 0}, {target->width, target->height}}; - offscreen_draws.push_back(clear); - continue; - } - if (draw.clear_flags) { - PreparedDraw clear{}; - clear.clear_flags = draw.clear_flags; - const auto color = unpack_argb(draw.clear_color); - clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; - clear.clear_depth.depthStencil = {draw.clear_z, 0}; - // D3D8 Clear with no rectangles clears the current viewport. - const auto viewport = draw_viewport(draw); - const float x0 = std::clamp(viewport.x, 0.0f, float(extent_.width)); - const float y0 = std::clamp(viewport.y, 0.0f, float(extent_.height)); - const float x1 = std::clamp(viewport.x + viewport.width, 0.0f, float(extent_.width)); - const float y1 = std::clamp(viewport.y + viewport.height, 0.0f, float(extent_.height)); - clear.clear_rect = {{std::int32_t(x0), std::int32_t(y0)}, - {std::uint32_t(x1 - x0), std::uint32_t(y1 - y0)}}; - prepared_draws.push_back(clear); - continue; - } - if (draw.positions.empty() || draw.indices.empty()) continue; - const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw); - const GeometryId id{draw.geometry_key, signature}; - auto [it, inserted] = geometries_.try_emplace(id); - auto& geometry = it->second; - if (inserted) { - upload_geometry(draw, geometry); - } else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) { - throw std::runtime_error("geometry key/revision reused with a different size"); - } - geometry.last_used_frame = frame_number_; - if (draw.geometry_key) active_keys.insert(draw.geometry_key); - - float skin_offset_encoded = 0.0f; - if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && f_->bone_mapped) { - const auto bone_count = static_cast(draw.bone_matrices.size() / 16); - if (std::size_t(bone_cursor) + bone_count <= f_->bone_capacity) { - std::memcpy( - f_->bone_mapped + std::size_t(bone_cursor) * 16, - draw.bone_matrices.data(), - draw.bone_matrices.size() * sizeof(float)); - skin_offset_encoded = float(bone_cursor + 1u); - bone_cursor += bone_count; - ++skinned_draw_count; - } else throw std::runtime_error("bone palette capacity exceeded"); - } - - const bool is_alpha = draw.alpha_blend != 0; - std::uint8_t cull = 0; - // D3D8 cull mode names describe the winding to discard. With the - // shader's Y flip, clockwise is Vulkan's front-facing winding. - if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise - else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise - const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0; - std::uint8_t blend = 0; - if (is_alpha) { - if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 && - draw.dest_blend >= 1 && draw.dest_blend <= 11) { - blend = static_cast((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4)); - } else { - blend = 1; - } - } - const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, - static_cast(draw.z_func), target != nullptr); - // D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples. - const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1; - const auto descriptor = get_texture_descriptor( - draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw); - - auto constants = make_push_constants(draw, skin_offset_encoded); - if (draw.pretransformed) { - const float screen_width = float(logical_width()); - const float screen_height = float(logical_height()); - // Screen pixels (y down) to D3D NDC (y up); the shader's Y flip then maps to Vulkan. - constants.mvp = {2.0f / screen_width, 0, 0, 0, - 0, -2.0f / screen_height, 0, 0, - 0, 0, 1, 0, - -1, 1, 0, 1}; - } - PreparedDraw prepared_draw{}; - prepared_draw.target = target; - prepared_draw.geometry = &geometry; - prepared_draw.pipeline = pipeline; - prepared_draw.descriptor = descriptor; - prepared_draw.constants = constants; - prepared_draw.fixed_state = make_fixed_function_state(draw); - // XYZRHW vertices are already in screen space and bypass the viewport transform. - prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw); - if (draw.pretransformed && draw.ui_offset_x != 0) - prepared_draw.viewport.x += draw.ui_offset_x * float(extent_.width) / float(logical_width()); - if (target) { - // D3DVIEWPORT8 in the target's own pixels. - prepared_draw.viewport = draw.viewport[2] > 0 && draw.viewport[3] > 0 - ? VkViewport{draw.viewport[0], draw.viewport[1], draw.viewport[2], draw.viewport[3], - draw.viewport_z[0], draw.viewport_z[1]} - : VkViewport{0, 0, float(target->width), float(target->height), 0, 1}; - offscreen_draws.push_back(prepared_draw); - } else prepared_draws.push_back(prepared_draw); - vertex_count += geometry.vertex_count; - index_count += geometry.index_count; - } - for (auto it = geometries_.begin(); it != geometries_.end();) { - const auto age = frame_number_ - it->second.last_used_frame; - const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key); - if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) { - release_geometry(it->second); - it = geometries_.erase(it); - } else ++it; - } - - std::vector ui_batches; - std::size_t ui_quad_count = 0; - if (f_->ui_mapped) { - const uint32_t target_ui_w = ui_width ? ui_width : 960; - const uint32_t target_ui_h = ui_height ? ui_height : 640; - update_fps_counter(); - if (touch_controller.is_enabled() || show_fps_) { - std::vector combined_ui = ui_commands; - if (touch_controller.is_enabled()) { - touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h), ui_safe_inset()); - touch_controller.append_ui_commands(combined_ui); - } - if (show_fps_ && fps_ >= 0) { - // Top left, inside the safe area and clear of 40250's corner icon. - const float inset = float(std::min(ui_safe_inset(), int(target_ui_w) / 8)); - TouchController::draw_segment_text(combined_ui, "FPS " + std::to_string(fps_), - inset + 40.0f, 8.0f, 12.0f, 0xFF40FF40); - } - if (!combined_ui.empty()) { - build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count); - } - } else if (!ui_commands.empty()) { - build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count); - } - } - - ensure_state_capacity(prepared_draws.size() + offscreen_draws.size() + ui_batches.size()); - std::size_t state_index = 0; - for (auto& draw : offscreen_draws) { - if (!draw.geometry) continue; - draw.state_offset = static_cast(state_index * state_stride_); - std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); - ++state_index; - } - last_offscreen_draw_count_ = offscreen_draws.size(); - for (auto& draw : prepared_draws) { - if (!draw.geometry) continue; - draw.state_offset = static_cast(state_index * state_stride_); - std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); - ++state_index; - } - const FixedFunctionState ui_state{}; - for (auto& batch : ui_batches) { - batch.state_offset = static_cast(state_index * state_stride_); - std::memcpy(f_->state_mapped + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); - ++state_index; - } - - const auto prepared = std::chrono::steady_clock::now(); - f_->recording = false; - check(vkResetFences(device_, 1, &f_->fence), "vkResetFences"); - - VkClearValue clears[3]{}; - // 40250 clears the back buffer to 0xff000000 once at device creation and afterwards - // only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer). - clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}}; - clears[1].depthStencil = {1.0f, 0}; - VkPipeline current_pipeline = VK_NULL_HANDLE; - VkDescriptorSet current_descriptor = VK_NULL_HANDLE; - - VkViewport current_viewport{-1, -1, -1, -1, -1, -1}; - auto set_viewport = [&](const VkViewport& viewport) { - if (std::memcmp(&viewport, ¤t_viewport, sizeof(viewport)) == 0) return; - vkCmdSetViewport(f_->command, 0, 1, &viewport); - current_viewport = viewport; - }; - - // One prepared draw or Clear inside the current render pass. - auto record_prepared = [&](const PreparedDraw& prepared_draw) { - if (!prepared_draw.geometry) { - VkClearAttachment attachments[2]{}; - std::uint32_t attachment_count = 0; - if (prepared_draw.clear_flags & 1u) { // D3DCLEAR_TARGET - attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; - attachments[attachment_count].colorAttachment = 0; - attachments[attachment_count].clearValue = prepared_draw.clear_color; - ++attachment_count; - } - if (prepared_draw.clear_flags & 2u) { // D3DCLEAR_ZBUFFER - attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; - attachments[attachment_count].clearValue = prepared_draw.clear_depth; - ++attachment_count; - } - const VkClearRect rect{prepared_draw.clear_rect, 0, 1}; - if (attachment_count && rect.rect.extent.width && rect.rect.extent.height) - vkCmdClearAttachments(f_->command, attachment_count, attachments, 1, &rect); - return; - } - set_viewport(prepared_draw.viewport); - if (prepared_draw.pipeline != current_pipeline) { - vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); - current_pipeline = prepared_draw.pipeline; - } - if (prepared_draw.descriptor != current_descriptor) { - vkCmdBindDescriptorSets( - f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); - current_descriptor = prepared_draw.descriptor; - } - vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, - &f_->bone_descriptor_set, 1, &prepared_draw.state_offset); - const auto& geometry = *prepared_draw.geometry; - const VkDeviceSize vertex_offset = 0; - vkCmdBindVertexBuffers(f_->command, 0, 1, &geometry.buffer, &vertex_offset); - vkCmdBindIndexBuffer(f_->command, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); - vkCmdPushConstants( - f_->command, - layout_, - VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, - 0, - sizeof(prepared_draw.constants), - &prepared_draw.constants); - vkCmdDrawIndexed(f_->command, geometry.index_count, 1, 0, 0, 0); - }; - - // Render targets first used this frame (drawn or only sampled) start cleared. - for (auto& entry : render_targets_) - if (!entry.second.initialized) initialize_render_target(entry.second); - // Offscreen passes: one per run of draws into the same target. - for (std::size_t i = 0; i < offscreen_draws.size();) { - RenderTarget* target = offscreen_draws[i].target; - VkRenderPassBeginInfo offscreen_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; - offscreen_begin.renderPass = offscreen_pass_; - offscreen_begin.framebuffer = target->framebuffer; - offscreen_begin.renderArea.extent = {target->width, target->height}; - vkCmdBeginRenderPass(f_->command, &offscreen_begin, VK_SUBPASS_CONTENTS_INLINE); - for (; i < offscreen_draws.size() && offscreen_draws[i].target == target; ++i) - record_prepared(offscreen_draws[i]); - vkCmdEndRenderPass(f_->command); - } - - VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; - pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); - pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears; - vkCmdBeginRenderPass(f_->command, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); - - auto record_ui_pass = [&](bool behind_3d) { - bool ui_bound = false; - set_viewport(full_viewport()); - for (const auto& batch : ui_batches) { - if (batch.behind_3d != behind_3d) continue; - if (!ui_bound) { - const VkDeviceSize v_offset = 0; - vkCmdBindVertexBuffers(f_->command, 0, 1, &f_->ui_buffer, &v_offset); - vkCmdBindIndexBuffer(f_->command, f_->ui_buffer, f_->ui_vertex_bytes, VK_INDEX_TYPE_UINT32); - ui_bound = true; - } - if (batch.pipeline != current_pipeline) { - vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); - current_pipeline = batch.pipeline; - } - if (batch.descriptor != current_descriptor) { - vkCmdBindDescriptorSets( - f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); - current_descriptor = batch.descriptor; - } - vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, - &f_->bone_descriptor_set, 1, &batch.state_offset); - vkCmdPushConstants( - f_->command, - layout_, - VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, - 0, - sizeof(batch.constants), - &batch.constants); - vkCmdDrawIndexed(f_->command, batch.index_count, 1, batch.first_index, 0, 0); - } - }; - - record_ui_pass(true); - - for (const auto& prepared_draw : prepared_draws) record_prepared(prepared_draw); - - record_ui_pass(false); - - vkCmdEndRenderPass(f_->command); - VkBuffer readback = VK_NULL_HANDLE; - VkDeviceMemory readback_memory = VK_NULL_HANDLE; - const bool screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_; - if (screenshot) { - const VkDeviceSize bytes = VkDeviceSize(extent_.width) * extent_.height * 4; - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = bytes; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; - check(vkCreateBuffer(device_, &info, nullptr, &readback), "vkCreateBuffer readback"); - VkMemoryRequirements requirements{}; - vkGetBufferMemoryRequirements(device_, readback, &requirements); - VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - allocation.allocationSize = requirements.size; - allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, - VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &allocation, nullptr, &readback_memory), "vkAllocateMemory readback"); - check(vkBindBufferMemory(device_, readback, readback_memory, 0), "vkBindBufferMemory readback"); - VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; - barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.oldLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.image = swapchain_images_.at(image_index); - barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, - VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - VkBufferImageCopy region{}; - region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; - region.imageExtent = {extent_.width, extent_.height, 1}; - vkCmdCopyImageToBuffer(f_->command, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, ®ion); - barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.dstAccessMask = 0; - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, - VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - } - // Debug: MT_DUMP_RT=prefix writes each render target of the screenshot frame to _.pgm. - struct RtDump { VkBuffer buffer; VkDeviceMemory memory; std::uint32_t w, h; }; - std::vector rt_dumps; - const char* dump_rt = std::getenv("MT_DUMP_RT"); - if (screenshot && dump_rt && *dump_rt) { - const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; - for (auto& entry : render_targets_) { - RenderTarget& rt = entry.second; - RtDump dump{VK_NULL_HANDLE, VK_NULL_HANDLE, rt.width, rt.height}; - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = VkDeviceSize(rt.width) * rt.height * texel; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; - check(vkCreateBuffer(device_, &info, nullptr, &dump.buffer), "vkCreateBuffer rt dump"); - VkMemoryRequirements requirements{}; - vkGetBufferMemoryRequirements(device_, dump.buffer, &requirements); - VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - allocation.allocationSize = requirements.size; - allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, - VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &allocation, nullptr, &dump.memory), "vkAllocateMemory rt dump"); - check(vkBindBufferMemory(device_, dump.buffer, dump.memory, 0), "vkBindBufferMemory rt dump"); - VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_SHADER_READ_BIT; - barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.image = rt.texture.image; - barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, - VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - VkBufferImageCopy region{}; - region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; - region.imageExtent = {rt.width, rt.height, 1}; - vkCmdCopyImageToBuffer(f_->command, rt.texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dump.buffer, 1, ®ion); - barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, - VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - rt_dumps.push_back(dump); - } - } - vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, f_->query_base + 1); - check(vkEndCommandBuffer(f_->command), "vkEndCommandBuffer"); - f_->has_pending_query = true; - - const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; - VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; - submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &f_->acquire; submit.pWaitDstStageMask = &stage; - submit.commandBufferCount = 1; submit.pCommandBuffers = &f_->command; - submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_[image_index]; - check(vkQueueSubmit(queue_, 1, &submit, f_->fence), "vkQueueSubmit"); - ++frame_slot_; - if (screenshot) { - check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences readback"); - void* mapped = nullptr; - check(vkMapMemory(device_, readback_memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback"); - write_bmp(screenshot_path_, static_cast(mapped)); - vkUnmapMemory(device_, readback_memory); - const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; - for (std::size_t n = 0; n < rt_dumps.size(); ++n) { - const auto& dump = rt_dumps[n]; - void* data = nullptr; - check(vkMapMemory(device_, dump.memory, 0, VK_WHOLE_SIZE, 0, &data), "vkMapMemory rt dump"); - const auto* bytes = static_cast(data); - std::ofstream out(std::string(dump_rt) + "_" + std::to_string(n) + ".pgm", std::ios::binary); - out << "P5\n" << dump.w << " " << dump.h << "\n255\n"; - for (std::size_t i = 0; i < std::size_t(dump.w) * dump.h; ++i) { - std::uint8_t g = texel == 2 ? std::uint8_t(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31) : bytes[i * 4]; - out.put(char(g)); - } - vkUnmapMemory(device_, dump.memory); - vkDestroyBuffer(device_, dump.buffer, nullptr); - vkFreeMemory(device_, dump.memory, nullptr); - } - vkDestroyBuffer(device_, readback, nullptr); - vkFreeMemory(device_, readback_memory, nullptr); - } - const auto submitted = std::chrono::steady_clock::now(); - VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; - present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_[image_index]; - present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; - const auto presented = vkQueuePresentKHR(queue_, &present); - // An Android surface can continuously report SUBOPTIMAL while the app deliberately - // uses the identity transform. The image is still presentable; rebuilding every frame - // stalls rendering without changing that status. - if (presented == VK_ERROR_OUT_OF_DATE_KHR) { - recreate_swapchain(); - } else if (presented != VK_SUCCESS && presented != VK_SUBOPTIMAL_KHR) { - check(presented, "vkQueuePresentKHR"); - } - const auto presented_at = std::chrono::steady_clock::now(); - const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); - last_sync_ms_ = std::chrono::duration(synchronized - start).count(); - timings_.sync_ms += last_sync_ms_; - timings_.fence_ms += std::chrono::duration(fenced - start).count(); - timings_.prepare_ms += prepare_ms; - if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; - timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); - timings_.present_ms += std::chrono::duration(presented_at - submitted).count(); - last_draw_count_ = prepared_draws.size(); - last_skinned_draw_count_ = skinned_draw_count; - last_ui_batch_count_ = ui_batches.size(); - last_ui_quad_count_ = ui_quad_count; - last_vertex_count_ = vertex_count; - last_index_count_ = index_count; - } - - std::size_t draw_count() const { return last_draw_count_; } - std::size_t skinned_draw_count() const { return last_skinned_draw_count_; } - std::size_t ui_batch_count() const { return last_ui_batch_count_; } - std::size_t ui_quad_count() const { return last_ui_quad_count_; } - std::size_t vertex_count() const { return last_vertex_count_; } - std::size_t index_count() const { return last_index_count_; } - std::size_t upload_count() const { return upload_count_; } - std::size_t uploaded_bytes() const { return uploaded_bytes_; } - std::size_t texture_upload_count() const { return texture_upload_count_; } - std::size_t texture_uploaded_bytes() const { return texture_uploaded_bytes_; } - const std::string& device_name() const { return device_name_; } - const char* present_mode_name() const { - switch (present_mode_) { - case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE"; - case VK_PRESENT_MODE_MAILBOX_KHR: return "MAILBOX"; - default: return "FIFO"; - } - } - Timings timings() const { return timings_; } - native_perf::RendererTotals perf_totals() const { - native_perf::RendererTotals t; - t.sync_ms = timings_.sync_ms; - t.fence_ms = timings_.fence_ms; - t.prepare_ms = timings_.prepare_ms; - t.geometry_upload_ms = timings_.geometry_upload_ms; - t.texture_upload_ms = timings_.texture_upload_ms; - t.submit_ms = timings_.submit_ms; - t.present_ms = timings_.present_ms; - t.gpu_ms = timings_.gpu_ms; - t.gpu_samples = timings_.gpu_samples; - t.uploads = upload_count_; - t.uploaded_bytes = uploaded_bytes_; - t.texture_uploads = texture_upload_count_; - t.texture_bytes = texture_uploaded_bytes_; - t.draws = last_draw_count_; - t.skinned_draws = last_skinned_draw_count_; - t.ui_batches = last_ui_batch_count_; - t.vertices = last_vertex_count_; - t.last_texture = &last_texture_upload_name_; - return t; - } - -private: - void recreate_swapchain() { - int w = 0, h = 0; - SDL_GetWindowSizeInPixels(window_, &w, &h); - if (w <= 0 || h <= 0) return; - vkDeviceWaitIdle(device_); - for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); - pipelines_.clear(); - for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr); - framebuffers_.clear(); - if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; } - if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; } - if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; } - destroy_msaa_color(); - for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); - image_views_.clear(); - if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; } - create_swapchain(); - create_framebuffers(); - get_pipeline(0, 2, 0); - } - - void collect_pending_gpu_timestamp(FrameSlot& slot) { - if (!slot.has_pending_query) return; - slot.has_pending_query = false; - std::uint64_t timestamps[2] = {}; - const VkResult res = vkGetQueryPoolResults( - device_, query_pool_, slot.query_base, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), - VK_QUERY_RESULT_64_BIT); - if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) { - const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_); - timings_.gpu_ms += ns * 1e-6; - ++timings_.gpu_samples; - } - } - - void build_ui_batches( - std::uint32_t ui_width, - std::uint32_t ui_height, - const std::vector& commands, - const std::unordered_map>& capture_textures, - std::vector& batches, - std::size_t& out_quad_count) { - ensure_ui_capacity(commands.size()); - auto* vertices = reinterpret_cast(f_->ui_mapped); - auto* indices = reinterpret_cast(f_->ui_mapped + f_->ui_vertex_bytes); - std::uint32_t v_count = 0; - std::uint32_t i_count = 0; - const float inv_w = 2.0f / float(ui_width); - const float inv_h = 2.0f / float(ui_height); - - for (const auto& cmd : commands) { - if (v_count + 4 > f_->ui_vertex_capacity || i_count + 6 > f_->ui_index_capacity) - throw std::runtime_error("UI command buffer capacity exceeded"); - if (cmd.kind == UIRenderCommand::Text) continue; - - float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2; - float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev; - const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1; - - float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; - std::array top_color = unpack_argb(cmd.argb); - std::array bot_color = - cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color; - if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue; - - if (cmd.kind == UIRenderCommand::Line) { - const float dx = x2 - x1, dy = y2 - y1; - const float len = std::sqrt(dx * dx + dy * dy); - if (len < 0.1f) continue; - const float nx = -dy / len * 0.5f; - const float ny = dx / len * 0.5f; - qx[0] = x1 + nx; qy[0] = y1 + ny; - qx[1] = x2 + nx; qy[1] = y2 + ny; - qx[2] = x1 - nx; qy[2] = y1 - ny; - qx[3] = x2 - nx; qy[3] = y2 - ny; - for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; } - } else if (cmd.kind == UIRenderCommand::Image && cmd.quad) { - const bool axis_aligned = - std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f && - std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f && - std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f && - std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f; - for (int k = 0; k < 4; ++k) { - mu[k] = cmd.mu[k]; - mv[k] = cmd.mv[k]; - } - if (has_clip && axis_aligned) { - const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3]; - const float cx1 = std::max(ox1, cmd.clip_x1); - const float cy1 = std::max(oy1, cmd.clip_y1); - const float cx2 = std::min(ox2, cmd.clip_x2); - const float cy2 = std::min(oy2, cmd.clip_y2); - if (cx2 <= cx1 || cy2 <= cy1) continue; - float msu = cmd.mu[0], meu = cmd.mu[1]; - float msv = cmd.mv[0], mev = cmd.mv[2]; - if (ox2 > ox1) { - const float t0 = (cx1 - ox1) / (ox2 - ox1); - const float t1 = (cx2 - ox1) / (ox2 - ox1); - const float du = eu - su; - su = cmd.su + du * t0; - eu = cmd.su + du * t1; - const float mdu = meu - msu; - msu = cmd.mu[0] + mdu * t0; - meu = cmd.mu[0] + mdu * t1; - } - if (oy2 > oy1) { - const float t0 = (cy1 - oy1) / (oy2 - oy1); - const float t1 = (cy2 - oy1) / (oy2 - oy1); - const float dv = ev - sv; - sv = cmd.sv + dv * t0; - ev = cmd.sv + dv * t1; - const float mdv = mev - msv; - msv = cmd.mv[0] + mdv * t0; - mev = cmd.mv[0] + mdv * t1; - } - qx[0] = cx1; qy[0] = cy1; - qx[1] = cx2; qy[1] = cy1; - qx[2] = cx1; qy[2] = cy2; - qx[3] = cx2; qy[3] = cy2; - mu[0] = msu; mv[0] = msv; - mu[1] = meu; mv[1] = msv; - mu[2] = msu; mv[2] = mev; - mu[3] = meu; mv[3] = mev; - } else { - if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) - continue; - for (int k = 0; k < 4; ++k) { - qx[k] = cmd.qx[k]; - qy[k] = cmd.qy[k]; - } - } - u[0] = su; v[0] = sv; - u[1] = eu; v[1] = sv; - u[2] = su; v[2] = ev; - u[3] = eu; v[3] = ev; - } else if (cmd.kind == UIRenderCommand::Bar && cmd.quad) { - // An untextured polygon piece (CPythonGraphic::RenderCoolTimeBox's fan triangles). - if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) - continue; - for (int k = 0; k < 4; ++k) { - qx[k] = cmd.qx[k]; - qy[k] = cmd.qy[k]; - u[k] = 0.0f; - v[k] = 0.0f; - } - } else { - if (has_clip) { - x1 = std::max(x1, cmd.clip_x1); - y1 = std::max(y1, cmd.clip_y1); - x2 = std::min(x2, cmd.clip_x2); - y2 = std::min(y2, cmd.clip_y2); - } - if (x2 <= x1 || y2 <= y1) continue; - qx[0] = x1; qy[0] = y1; - qx[1] = x2; qy[1] = y1; - qx[2] = x1; qy[2] = y2; - qx[3] = x2; qy[3] = y2; - u[0] = su; v[0] = sv; - u[1] = eu; v[1] = sv; - u[2] = su; v[2] = ev; - u[3] = eu; v[3] = ev; - } - - const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string(); - const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string(); - const bool is_masked = !mask_name.empty(); - // CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use - // INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR. - const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image - ? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1)) - : 1; - const VkPipeline pipeline = get_pipeline(0, 0, blend_mode); - const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures); - - for (int k = 0; k < 4; ++k) { - Vertex& vert = vertices[v_count + k]; - vert.position[0] = qx[k] * inv_w - 1.0f; - vert.position[1] = 1.0f - qy[k] * inv_h; - vert.position[2] = 0.0f; - vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f; - vert.uv[0] = u[k]; vert.uv[1] = v[k]; - const auto& col = (k < 2) ? top_color : bot_color; - vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3]; - vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0; - vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f; - vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k]; - vert.rhw = 1.0f; - } - indices[i_count + 0] = v_count + 0; - indices[i_count + 1] = v_count + 1; - indices[i_count + 2] = v_count + 2; - indices[i_count + 3] = v_count + 2; - indices[i_count + 4] = v_count + 1; - indices[i_count + 3 + 2] = v_count + 3; - - const float mask_flag = is_masked ? 1.0f : 0.0f; - if (!batches.empty() && - batches.back().behind_3d == cmd.behind_3d && - batches.back().pipeline == pipeline && - batches.back().descriptor == descriptor && - batches.back().constants.params[2] == mask_flag) { - batches.back().index_count += 6; - } else { - UiBatch batch{}; - batch.first_index = i_count; - batch.index_count = 6; - batch.pipeline = pipeline; - batch.descriptor = descriptor; - batch.behind_3d = cmd.behind_3d; - batch.constants.mvp = kIdentityMatrix; - batch.constants.params = {0.0f, 0.0f, mask_flag, 0.0f}; - batches.push_back(batch); - } - v_count += 4; - i_count += 6; - ++out_quad_count; - } - } - - std::uint32_t find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const { - VkPhysicalDeviceMemoryProperties properties{}; - vkGetPhysicalDeviceMemoryProperties(physical_, &properties); - for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i) - if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags) - return i; - throw std::runtime_error("no matching Vulkan memory type"); - } - - void select_device() { - std::uint32_t count = 0; - check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count"); - if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK"); - std::vector candidates(count); - check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices"); - for (auto candidate : candidates) { - std::uint32_t family_count = 0; - vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr); - std::vector families(family_count); - vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data()); - for (std::uint32_t i = 0; i < family_count; ++i) { - VkBool32 present = VK_FALSE; - check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support"); - if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) { - physical_ = candidate; queue_family_ = i; break; - } - } - if (physical_) break; - } - if (!physical_) throw std::runtime_error("no graphics/present queue family"); - VkPhysicalDeviceProperties properties{}; - vkGetPhysicalDeviceProperties(physical_, &properties); - device_name_ = properties.deviceName; - timestamp_period_ns_ = properties.limits.timestampPeriod; - VkPhysicalDeviceFeatures supported_features{}; - vkGetPhysicalDeviceFeatures(physical_, &supported_features); - VkPhysicalDeviceFeatures enabled_features{}; - if (supported_features.samplerAnisotropy) { - enabled_features.samplerAnisotropy = VK_TRUE; - max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy); - } - // Multisampled 3D edges; MT_MSAA=1 keeps the single-sample path, 2/4/8 request a count. - { - int wanted = 4; - if (const char* env = std::getenv("MT_MSAA")) wanted = std::max(1, std::atoi(env)); - const VkSampleCountFlags supported = - properties.limits.framebufferColorSampleCounts & properties.limits.framebufferDepthSampleCounts; - samples_ = VK_SAMPLE_COUNT_1_BIT; - for (VkSampleCountFlagBits c : {VK_SAMPLE_COUNT_8_BIT, VK_SAMPLE_COUNT_4_BIT, VK_SAMPLE_COUNT_2_BIT}) - if (int(c) <= wanted && (supported & c)) { samples_ = c; break; } - } - std::uint32_t extension_count = 0; - check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count"); - std::vector available(extension_count); - check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions"); - std::vector extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME}; - constexpr const char* portability_subset = "VK_KHR_portability_subset"; - for (const auto& extension : available) - if (std::strcmp(extension.extensionName, portability_subset) == 0) - extensions.push_back(portability_subset); - const float priority = 1.0f; - VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; - queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority; - VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; - device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info; - device_info.enabledExtensionCount = static_cast(extensions.size()); - device_info.ppEnabledExtensionNames = extensions.data(); - device_info.pEnabledFeatures = &enabled_features; - check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice"); - vkGetDeviceQueue(device_, queue_family_, 0, &queue_); - } - - void create_swapchain() { - VkSurfaceCapabilitiesKHR capabilities{}; - check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities"); - std::uint32_t count = 0; - check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count"); - if (!count) throw std::runtime_error("surface has no formats"); - std::vector formats(count); - check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats"); - VkSurfaceFormatKHR format = formats.front(); - for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate; - swapchain_format_ = format.format; - int width = 0, height = 0; - SDL_GetWindowSizeInPixels(window_, &width, &height); - if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent; - else { - extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width); - extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height); - } - - present_mode_ = VK_PRESENT_MODE_FIFO_KHR; - if (!vsync_) { - std::uint32_t mode_count = 0; - check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count"); - std::vector modes(mode_count); - if (mode_count) - check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes"); - if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end()) - present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR; - else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end()) - present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR; - } - - VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR}; - info.surface = surface_; - info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX); - info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_; - info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | - (capabilities.supportedUsageFlags & VK_IMAGE_USAGE_TRANSFER_SRC_BIT); - info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; - // Android surfaces can report a quarter-turn transform even when SDL's window is - // already landscape. Rendering in SDL window coordinates and requesting that transform - // again rotates the entire game image on presentation. - info.preTransform = (capabilities.supportedTransforms & VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR) - ? VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR : capabilities.currentTransform; -#ifdef __ANDROID__ - SDL_Log("swapchain: window=%dx%d extent=%ux%u currentTransform=%u preTransform=%u", - width, height, extent_.width, extent_.height, - unsigned(capabilities.currentTransform), unsigned(info.preTransform)); -#endif - info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_; - info.clipped = VK_TRUE; - check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR"); - check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count"); - std::vector images(count); - check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images"); - swapchain_images_ = images; - for (auto image : images) { - VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; - view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; - view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; - VkImageView handle = VK_NULL_HANDLE; - check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView"); - image_views_.push_back(handle); - } - VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - depth_info.imageType = VK_IMAGE_TYPE_2D; - depth_info.format = VK_FORMAT_D32_SFLOAT; - depth_info.extent = {extent_.width, extent_.height, 1}; - depth_info.mipLevels = 1; depth_info.arrayLayers = 1; - depth_info.samples = samples_; - depth_info.tiling = VK_IMAGE_TILING_OPTIMAL; - depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT; - depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth"); - VkMemoryRequirements depth_reqs{}; - vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs); - VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - depth_alloc.allocationSize = depth_reqs.size; - depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth"); - check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth"); - VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D; - depth_view.format = VK_FORMAT_D32_SFLOAT; - depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; - depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1; - check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth"); - if (samples_ != VK_SAMPLE_COUNT_1_BIT) create_msaa_color(); - } - - // The multisampled colour target the subpass resolves into the swapchain image. It never - // leaves the render pass, so it is transient (lazily allocated tile memory where available). - void create_msaa_color() { - VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - info.imageType = VK_IMAGE_TYPE_2D; - info.format = swapchain_format_; - info.extent = {extent_.width, extent_.height, 1}; - info.mipLevels = 1; info.arrayLayers = 1; - info.samples = samples_; - info.tiling = VK_IMAGE_TILING_OPTIMAL; - info.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSIENT_ATTACHMENT_BIT; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &info, nullptr, &msaa_image_), "vkCreateImage msaa color"); - VkMemoryRequirements reqs{}; - vkGetImageMemoryRequirements(device_, msaa_image_, &reqs); - VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - alloc.allocationSize = reqs.size; - try { - alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_LAZILY_ALLOCATED_BIT); - } catch (const std::runtime_error&) { - alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - } - check(vkAllocateMemory(device_, &alloc, nullptr, &msaa_memory_), "vkAllocateMemory msaa color"); - check(vkBindImageMemory(device_, msaa_image_, msaa_memory_, 0), "vkBindImageMemory msaa color"); - VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view.image = msaa_image_; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; - view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; - view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; - check(vkCreateImageView(device_, &view, nullptr, &msaa_view_), "vkCreateImageView msaa color"); - } - - void destroy_msaa_color() { - if (msaa_view_) { vkDestroyImageView(device_, msaa_view_, nullptr); msaa_view_ = VK_NULL_HANDLE; } - if (msaa_image_) { vkDestroyImage(device_, msaa_image_, nullptr); msaa_image_ = VK_NULL_HANDLE; } - if (msaa_memory_) { vkFreeMemory(device_, msaa_memory_, nullptr); msaa_memory_ = VK_NULL_HANDLE; } - } - - void create_framebuffers() { - for (auto view : image_views_) { - // Single-sample: {swapchain, depth}. MSAA: {msaa colour, depth, swapchain resolve}. - VkImageView fb_attachments[3] = {view, depth_view_, VK_NULL_HANDLE}; - std::uint32_t count = 2; - if (samples_ != VK_SAMPLE_COUNT_1_BIT) { - fb_attachments[0] = msaa_view_; - fb_attachments[2] = view; - count = 3; - } - VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; - framebuffer.renderPass = pass_; framebuffer.attachmentCount = count; framebuffer.pAttachments = fb_attachments; - framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; - VkFramebuffer handle = VK_NULL_HANDLE; - check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer"); - framebuffers_.push_back(handle); - } - } - - void create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) { - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = size; - info.usage = usage; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host"); - VkMemoryRequirements reqs{}; - vkGetBufferMemoryRequirements(device_, buffer, &reqs); - VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - alloc.allocationSize = reqs.size; - alloc.memoryTypeIndex = find_memory_type( - reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host"); - check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host"); - check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host"); - } - - void create_descriptors_and_buffers() { - VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; - sampler_info.magFilter = VK_FILTER_LINEAR; - sampler_info.minFilter = VK_FILTER_LINEAR; - sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; - sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT; - sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT; - sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; - sampler_info.minLod = 0.0f; - sampler_info.maxLod = VK_LOD_CLAMP_NONE; - if (max_anisotropy_ > 1.0f) { - sampler_info.anisotropyEnable = VK_TRUE; - sampler_info.maxAnisotropy = max_anisotropy_; - } - check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler"); - - VkSamplerCreateInfo clamp_info = sampler_info; - clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp"); - - VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; - ui_sampler_info.magFilter = VK_FILTER_LINEAR; - ui_sampler_info.minFilter = VK_FILTER_LINEAR; - ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST; - ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - ui_sampler_info.minLod = 0.0f; - ui_sampler_info.maxLod = 0.0f; - ui_sampler_info.anisotropyEnable = VK_FALSE; - ui_sampler_info.maxAnisotropy = 1.0f; - check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui"); - - VkDescriptorSetLayoutBinding tex_bindings[2]{}; - tex_bindings[0].binding = 0; - tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - tex_bindings[0].descriptorCount = 1; - tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; - tex_bindings[1].binding = 1; - tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - tex_bindings[1].descriptorCount = 1; - tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; - VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; - layout_info.bindingCount = 2; - layout_info.pBindings = tex_bindings; - check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex"); - - VkDescriptorSetLayoutBinding bone_bindings[2]{}; - bone_bindings[0].binding = 0; - bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; - bone_bindings[0].descriptorCount = 1; - bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT; - bone_bindings[1].binding = 1; - bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - bone_bindings[1].descriptorCount = 1; - bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; - VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; - bone_layout_info.bindingCount = 2; - bone_layout_info.pBindings = bone_bindings; - check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone"); - - VkDescriptorPoolSize pool_sizes[3] = { - {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, - {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4 * kMaxFramesInFlight}, - {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4 * kMaxFramesInFlight}}; - VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; - pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; - pool_info.maxSets = 8192; - pool_info.poolSizeCount = 3; - pool_info.pPoolSizes = pool_sizes; - check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool"); - - const std::uint8_t white_pixel[4] = {255, 255, 255, 255}; - upload_texture(1, 1, white_pixel, fallback_texture_, false, false); - fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); - fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); - - VkPhysicalDeviceProperties properties{}; - vkGetPhysicalDeviceProperties(physical_, &properties); - const auto alignment = std::max( - 1, properties.limits.minStorageBufferOffsetAlignment); - state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; - staging_alignment_ = std::max(16, properties.limits.optimalBufferCopyOffsetAlignment); - - for (auto& slot : frames_) { - void* bone_raw = nullptr; - create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, slot.bone_buffer, slot.bone_memory, &bone_raw); - slot.bone_mapped = static_cast(bone_raw); - slot.bone_capacity = kMaxBonesPerFrame; - for (std::size_t i = 0; i < 16; ++i) slot.bone_mapped[i] = kIdentityMatrix[i]; - - slot.state_capacity = 256; - void* state_raw = nullptr; - create_host_buffer(slot.state_capacity * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, - slot.state_buffer, slot.state_memory, &state_raw); - slot.state_mapped = static_cast(state_raw); - - VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - bone_alloc.descriptorPool = descriptor_pool_; - bone_alloc.descriptorSetCount = 1; - bone_alloc.pSetLayouts = &bone_descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &bone_alloc, &slot.bone_descriptor_set), "vkAllocateDescriptorSets bone"); - - VkDescriptorBufferInfo buffer_infos[2] = { - {slot.bone_buffer, 0, kBoneBufferBytes}, - {slot.state_buffer, 0, sizeof(FixedFunctionState)}}; - VkWriteDescriptorSet writes[2]{}; - for (int binding = 0; binding < 2; ++binding) { - writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[binding].dstSet = slot.bone_descriptor_set; - writes[binding].dstBinding = static_cast(binding); - writes[binding].descriptorCount = 1; - writes[binding].descriptorType = binding == 0 - ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - writes[binding].pBufferInfo = &buffer_infos[binding]; - } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); - - void* ui_raw = nullptr; - create_host_buffer( - kUiBufferBytes, - VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, - slot.ui_buffer, - slot.ui_memory, - &ui_raw); - slot.ui_mapped = static_cast(ui_raw); - slot.ui_vertex_capacity = kMaxUiVerticesPerFrame; - slot.ui_index_capacity = kMaxUiIndicesPerFrame; - slot.ui_vertex_bytes = kUiVertexBytes; - slot.staging_wanted = kStagingRingBytes; - } - } - - // The slot's fence has signalled: its oversize staging buffers are free, and the ring restarts - // (grown to the largest frame's uploads seen so far, so texture streaming stays in the ring). - void release_retired_buffers(FrameSlot& slot) { - for (auto [buffer, memory] : slot.retired_buffers) { - vkDestroyBuffer(device_, buffer, nullptr); - vkFreeMemory(device_, memory, nullptr); - } - slot.retired_buffers.clear(); - } - - void reset_staging(FrameSlot& slot) { - release_retired_buffers(slot); - slot.staging_used = 0; - if (slot.staging_wanted <= slot.staging_capacity) return; - if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); - if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); - if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); - void* raw = nullptr; - slot.staging_capacity = std::min(slot.staging_wanted, kStagingRingMaxBytes); - create_host_buffer(slot.staging_capacity, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, - slot.staging_buffer, slot.staging_memory, &raw); - slot.staging_mapped = static_cast(raw); - slot.staging_wanted = slot.staging_capacity; - } - - void ensure_bone_capacity(std::size_t required) { - if (required <= f_->bone_capacity) return; - VkPhysicalDeviceProperties properties{}; - vkGetPhysicalDeviceProperties(physical_, &properties); - const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); - if (required > max_bones || required > UINT32_MAX) - throw std::runtime_error("bone palette exceeds device storage-buffer limit"); - std::size_t next = std::min(max_bones, std::max(required, f_->bone_capacity * 2)); - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - void* mapped = nullptr; - create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, f_->bone_memory); - vkDestroyBuffer(device_, f_->bone_buffer, nullptr); - vkFreeMemory(device_, f_->bone_memory, nullptr); - f_->bone_buffer = buffer; - f_->bone_memory = memory; - f_->bone_mapped = static_cast(mapped); - f_->bone_capacity = next; - VkDescriptorBufferInfo info{f_->bone_buffer, 0, next * 16 * sizeof(float)}; - VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; - write.dstSet = f_->bone_descriptor_set; - write.dstBinding = 0; - write.descriptorCount = 1; - write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; - write.pBufferInfo = &info; - vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); - } - - void ensure_state_capacity(std::size_t required) { - if (required <= f_->state_capacity) return; - const auto next = std::max(required, f_->state_capacity * 2); - if (next > UINT32_MAX / state_stride_) - throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - void* mapped = nullptr; - create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, - buffer, memory, &mapped); - vkUnmapMemory(device_, f_->state_memory); - vkDestroyBuffer(device_, f_->state_buffer, nullptr); - vkFreeMemory(device_, f_->state_memory, nullptr); - f_->state_buffer = buffer; - f_->state_memory = memory; - f_->state_mapped = static_cast(mapped); - f_->state_capacity = next; - VkDescriptorBufferInfo info{f_->state_buffer, 0, sizeof(FixedFunctionState)}; - VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; - write.dstSet = f_->bone_descriptor_set; - write.dstBinding = 1; - write.descriptorCount = 1; - write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - write.pBufferInfo = &info; - vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); - } - - void ensure_ui_capacity(std::size_t command_count) { - if (command_count <= f_->ui_vertex_capacity / 4 && command_count <= f_->ui_index_capacity / 6) return; - if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); - const std::size_t quads = std::max(command_count, f_->ui_vertex_capacity / 2); - const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); - const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); - if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) - throw std::runtime_error("UI geometry exceeds 128 MiB safety limit"); - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - void* mapped = nullptr; - create_host_buffer(vertex_bytes + index_bytes, - VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, f_->ui_memory); - vkDestroyBuffer(device_, f_->ui_buffer, nullptr); - vkFreeMemory(device_, f_->ui_memory, nullptr); - f_->ui_buffer = buffer; - f_->ui_memory = memory; - f_->ui_mapped = static_cast(mapped); - f_->ui_vertex_capacity = quads * 4; - f_->ui_index_capacity = quads * 6; - f_->ui_vertex_bytes = vertex_bytes; - } - - void create_render_pass_and_layout() { - const bool msaa = samples_ != VK_SAMPLE_COUNT_1_BIT; - VkAttachmentDescription attachments[3]{}; - attachments[0].format = swapchain_format_; attachments[0].samples = samples_; - attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; - attachments[0].storeOp = msaa ? VK_ATTACHMENT_STORE_OP_DONT_CARE : VK_ATTACHMENT_STORE_OP_STORE; - attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; - attachments[0].finalLayout = msaa ? VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL : VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - // MSAA resolve target: the swapchain image. - attachments[2].format = swapchain_format_; attachments[2].samples = VK_SAMPLE_COUNT_1_BIT; - attachments[2].loadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; attachments[2].storeOp = VK_ATTACHMENT_STORE_OP_STORE; - attachments[2].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[2].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = samples_; - attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; - attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; - attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - - VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; - VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; - VkAttachmentReference resolve_ref{2, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; - VkSubpassDescription subpass{}; - subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; - subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; - if (msaa) subpass.pResolveAttachments = &resolve_ref; - subpass.pDepthStencilAttachment = &depth_ref; - - VkSubpassDependency dependency{}; - dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0; - dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; - dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; - dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; - - VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; - pass_info.attachmentCount = msaa ? 3 : 2; pass_info.pAttachments = attachments; - pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; - pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency; - check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass"); - create_framebuffers(); - - const auto vert = read_spirv("native.vert.spv"); - const auto frag = read_spirv("native.frag.spv"); - auto module = [&](const std::vector& code) { - VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO}; - info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data(); - VkShaderModule handle = VK_NULL_HANDLE; - check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule"); - return handle; - }; - vertex_module_ = module(vert); - fragment_module_ = module(frag); - - VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_}; - VkPushConstantRange push_constants{}; - push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; - push_constants.size = sizeof(PushConstants); - VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO}; - layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts; - layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants; - check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); - - get_pipeline(0, 2, 0); - create_offscreen_pass(); - } - - // The render pass of offscreen render-target textures: colour (R5G6B5 like the D3D surface when - // the device can render to it) + D32 depth, both loaded and stored, colour left shader-readable. - void create_offscreen_pass() { - VkFormatProperties properties{}; - vkGetPhysicalDeviceFormatProperties(physical_, VK_FORMAT_R5G6B5_UNORM_PACK16, &properties); - const VkFormatFeatureFlags wanted = - VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT; - offscreen_format_ = (properties.optimalTilingFeatures & wanted) == wanted - ? VK_FORMAT_R5G6B5_UNORM_PACK16 : VK_FORMAT_R8G8B8A8_UNORM; - VkAttachmentDescription attachments[2]{}; - attachments[0].format = offscreen_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; - attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; - attachments[0].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; - attachments[0].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; - attachments[0].initialLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - attachments[0].finalLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; - attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_STORE; - attachments[1].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; - attachments[1].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; - attachments[1].initialLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; - VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; - VkSubpassDescription subpass{}; - subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; - subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; - subpass.pDepthStencilAttachment = &depth_ref; - VkSubpassDependency dependencies[2]{}; - // Earlier work on the queue -- the previous frame sampling the texture, the previous - // offscreen pass, the first-use clear -- completes before this pass writes it. - dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL; dependencies[0].dstSubpass = 0; - dependencies[0].srcStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | - VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | - VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT; - dependencies[0].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | - VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT; - dependencies[0].dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | - VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT; - dependencies[0].dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | - VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; - // The back-buffer pass samples the result. - dependencies[1].srcSubpass = 0; dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL; - dependencies[1].srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; - dependencies[1].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; - dependencies[1].dstStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT; - dependencies[1].dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; - pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; - pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; - pass_info.dependencyCount = 2; pass_info.pDependencies = dependencies; - check(vkCreateRenderPass(device_, &pass_info, nullptr, &offscreen_pass_), "vkCreateRenderPass offscreen"); - } - - // "rt::x" -> its render target, created on first use (by a draw into it or by a draw - // sampling it); null for a malformed name. - RenderTarget* get_render_target(const std::string& name) { - if (auto it = render_targets_.find(name); it != render_targets_.end()) return &it->second; - unsigned id = 0, w = 0, h = 0; - if (std::sscanf(name.c_str(), "rt:%u:%ux%u", &id, &w, &h) != 3 || !w || !h || w > 4096 || h > 4096) - return nullptr; - RenderTarget& target = render_targets_[name]; - target.width = w; - target.height = h; - auto make_image = [&](VkFormat format, VkImageUsageFlags usage, VkImageAspectFlags aspect, - VkImage& image, VkDeviceMemory& memory, VkImageView& view) { - VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - info.imageType = VK_IMAGE_TYPE_2D; - info.format = format; - info.extent = {w, h, 1}; - info.mipLevels = 1; info.arrayLayers = 1; - info.samples = VK_SAMPLE_COUNT_1_BIT; - info.tiling = VK_IMAGE_TILING_OPTIMAL; - info.usage = usage; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &info, nullptr, &image), "vkCreateImage render target"); - VkMemoryRequirements reqs{}; - vkGetImageMemoryRequirements(device_, image, &reqs); - VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - alloc.allocationSize = reqs.size; - alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory render target"); - check(vkBindImageMemory(device_, image, memory, 0), "vkBindImageMemory render target"); - VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view_info.image = image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = format; - view_info.subresourceRange = {aspect, 0, 1, 0, 1}; - check(vkCreateImageView(device_, &view_info, nullptr, &view), "vkCreateImageView render target"); - }; - make_image(offscreen_format_, - VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | - VK_IMAGE_USAGE_TRANSFER_SRC_BIT, - VK_IMAGE_ASPECT_COLOR_BIT, target.texture.image, target.texture.memory, target.texture.view); - make_image(VK_FORMAT_D32_SFLOAT, - VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, - VK_IMAGE_ASPECT_DEPTH_BIT, target.depth_image, target.depth_memory, target.depth_view); - const VkImageView views[2] = {target.texture.view, target.depth_view}; - VkFramebufferCreateInfo fb{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; - fb.renderPass = offscreen_pass_; - fb.attachmentCount = 2; fb.pAttachments = views; - fb.width = w; fb.height = h; fb.layers = 1; - check(vkCreateFramebuffer(device_, &fb, nullptr, &target.framebuffer), "vkCreateFramebuffer render target"); - target.texture.descriptor = allocate_texture_descriptor_set(target.texture.view, fallback_texture_.view); - target.texture.ui_descriptor = allocate_ui_texture_descriptor_set(target.texture.view, fallback_texture_.view); - return ⌖ - } - - void release_render_target(RenderTarget& target) { - if (target.framebuffer) vkDestroyFramebuffer(device_, target.framebuffer, nullptr); - if (target.depth_view) vkDestroyImageView(device_, target.depth_view, nullptr); - if (target.depth_image) vkDestroyImage(device_, target.depth_image, nullptr); - if (target.depth_memory) vkFreeMemory(device_, target.depth_memory, nullptr); - release_texture(target.texture); - target = {}; - } - - // A new render target starts white with depth 1 (what the CPU surface's first Clear leaves), in - // the layouts offscreen_pass_ expects. Recorded outside any render pass. - void initialize_render_target(RenderTarget& target) { - VkImageMemoryBarrier barriers[2]{}; - for (auto& b : barriers) { - b.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER; - b.srcQueueFamilyIndex = b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - b.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; - b.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - b.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - } - barriers[0].image = target.texture.image; - barriers[0].subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - barriers[1].image = target.depth_image; - barriers[1].subresourceRange = {VK_IMAGE_ASPECT_DEPTH_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, - 0, 0, nullptr, 0, nullptr, 2, barriers); - const VkClearColorValue white{{1.0f, 1.0f, 1.0f, 1.0f}}; - const VkClearDepthStencilValue far{1.0f, 0}; - vkCmdClearColorImage(f_->command, target.texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &white, 1, - &barriers[0].subresourceRange); - vkCmdClearDepthStencilImage(f_->command, target.depth_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &far, 1, - &barriers[1].subresourceRange); - for (auto& b : barriers) { - b.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - b.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - } - barriers[0].newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - barriers[0].dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_READ_BIT; - barriers[1].newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - barriers[1].dstAccessMask = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, - VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | - VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, - 0, 0, nullptr, 0, nullptr, 2, barriers); - target.initialized = true; - } - - // offscreen: for offscreen_pass_ (single-sampled render-target texture) instead of pass_. - VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, - bool lines = false, std::uint8_t z_func = 4, bool offscreen = false) { - const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | - (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | - (std::uint32_t(z_func & 0xfu) << 25) | (offscreen ? (1u << 29) : 0u); - if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; - - VkPipelineShaderStageCreateInfo stages[2]{}; - stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; - stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main"; - stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main"; - - VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX}; - VkVertexInputAttributeDescription attributes[9] = { - {0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)}, - {1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)}, - {2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)}, - {3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)}, - {4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)}, - {5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)}, - {6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)}, - {7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)}, - {8, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, vertex_fog)}}; - VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO}; - input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding; - input.vertexAttributeDescriptionCount = 9; input.pVertexAttributeDescriptions = attributes; - - VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO}; - assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; - - VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; - // An offscreen target is at most the device's 4096 (GetDeviceCaps); the viewport clips inside it. - VkRect2D scissor{{0, 0}, offscreen ? VkExtent2D{4096, 4096} : extent_}; - VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; - viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; - viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; - // D3DVIEWPORT8 changes per draw (SetViewport), so the viewport is dynamic state. - const VkDynamicState dynamic_states[] = {VK_DYNAMIC_STATE_VIEWPORT}; - VkPipelineDynamicStateCreateInfo dynamic_info{VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO}; - dynamic_info.dynamicStateCount = 1; dynamic_info.pDynamicStates = dynamic_states; - - VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO}; - raster.polygonMode = VK_POLYGON_MODE_FILL; - raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE); - raster.frontFace = VK_FRONT_FACE_CLOCKWISE; - raster.lineWidth = 1; - - VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; - multisample.rasterizationSamples = offscreen ? VK_SAMPLE_COUNT_1_BIT : samples_; - - VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; - depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; - depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE; - switch (z_func) { - case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break; - case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break; - case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break; - case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break; - case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break; - case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break; - case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break; - default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break; - } - - auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor { - switch (d3d) { - case 1: return VK_BLEND_FACTOR_ZERO; - case 2: return VK_BLEND_FACTOR_ONE; - case 3: return VK_BLEND_FACTOR_SRC_COLOR; - case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR; - case 5: return VK_BLEND_FACTOR_SRC_ALPHA; - case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; - case 7: return VK_BLEND_FACTOR_DST_ALPHA; - case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; - case 9: return VK_BLEND_FACTOR_DST_COLOR; - case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR; - case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE; - default: return fallback; - } - }; - - VkPipelineColorBlendAttachmentState blend_attachment{}; - blend_attachment.colorWriteMask = 0xf; - if (blend > 0) { - blend_attachment.blendEnable = VK_TRUE; - if (blend >= 16) { - blend_attachment.srcColorBlendFactor = - d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA); - blend_attachment.dstColorBlendFactor = - d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA); - } else { - blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA; - blend_attachment.dstColorBlendFactor = - blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; - } - blend_attachment.colorBlendOp = VK_BLEND_OP_ADD; - // D3D8 has no separate alpha blend: SRCBLEND/DESTBLEND apply to alpha as well. - auto alpha_factor = [](VkBlendFactor factor) { - switch (factor) { - case VK_BLEND_FACTOR_SRC_COLOR: return VK_BLEND_FACTOR_SRC_ALPHA; - case VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; - case VK_BLEND_FACTOR_DST_COLOR: return VK_BLEND_FACTOR_DST_ALPHA; - case VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; - case VK_BLEND_FACTOR_SRC_ALPHA_SATURATE: return VK_BLEND_FACTOR_ONE; - default: return factor; - } - }; - blend_attachment.srcAlphaBlendFactor = alpha_factor(blend_attachment.srcColorBlendFactor); - blend_attachment.dstAlphaBlendFactor = alpha_factor(blend_attachment.dstColorBlendFactor); - blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD; - } - VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO}; - blending.attachmentCount = 1; blending.pAttachments = &blend_attachment; - - VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO}; - pipeline_info.stageCount = 2; pipeline_info.pStages = stages; - pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly; - pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster; - pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; - pipeline_info.pColorBlendState = &blending; - pipeline_info.pDynamicState = &dynamic_info; - pipeline_info.layout = layout_; pipeline_info.renderPass = offscreen ? offscreen_pass_ : pass_; - VkPipeline handle = VK_NULL_HANDLE; - check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); - pipelines_.emplace(key, handle); - return handle; - } - - VkDescriptorSet allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, - VkSampler sampler0 = VK_NULL_HANDLE, - VkSampler sampler1 = VK_NULL_HANDLE) { - VkDescriptorSet set = VK_NULL_HANDLE; - VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - desc_alloc.descriptorPool = descriptor_pool_; - desc_alloc.descriptorSetCount = 1; - desc_alloc.pSetLayouts = &descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets"); - - VkDescriptorImageInfo image_infos[2]{}; - image_infos[0].sampler = sampler0 ? sampler0 : sampler_; - image_infos[0].imageView = view0; - image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_); - image_infos[1].imageView = view1; - image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - - VkWriteDescriptorSet writes[2]{}; - for (int b = 0; b < 2; ++b) { - writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[b].dstSet = set; - writes[b].dstBinding = static_cast(b); - writes[b].descriptorCount = 1; - writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - writes[b].pImageInfo = &image_infos[b]; - } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); - return set; - } - - VkDescriptorSet allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) { - VkDescriptorSet set = VK_NULL_HANDLE; - VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - desc_alloc.descriptorPool = descriptor_pool_; - desc_alloc.descriptorSetCount = 1; - desc_alloc.pSetLayouts = &descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui"); - - VkDescriptorImageInfo image_infos[2]{}; - image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_; - image_infos[0].imageView = view0; - image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_); - image_infos[1].imageView = view1; - image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - - VkWriteDescriptorSet writes[2]{}; - for (int b = 0; b < 2; ++b) { - writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[b].dstSet = set; - writes[b].dstBinding = static_cast(b); - writes[b].descriptorCount = 1; - writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - writes[b].pImageInfo = &image_infos[b]; - } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); - return set; - } - - GpuTexture& get_gpu_texture( - const std::string& texture_name, - const std::unordered_map>& capture_textures) { - if (texture_name.empty()) return fallback_texture_; - if (texture_name.rfind("rt:", 0) == 0) { - RenderTarget* target = get_render_target(texture_name); - return target ? target->texture : fallback_texture_; - } - auto [it, inserted] = textures_.try_emplace(texture_name); - it->second.last_used_frame = frame_number_; - if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) - return it->second; - if (inserted) { - it->second.descriptor = fallback_texture_.descriptor; - it->second.ui_descriptor = fallback_texture_.ui_descriptor; - } - it->second.last_decode_attempt_frame = frame_number_; - - mtimage::Image decoded; - if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) { - decoded = decode_texture_bytes(raw->second.data(), raw->second.size()); - } -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (!decoded.ok()) { - if (texture_name.rfind("mem:", 0) == 0) { - UIMemoryTexture mem_tex; - if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { - const auto mtra = native_draw_capture::encode_raw_argb_as_mtra( - static_cast(mem_tex.width), - static_cast(mem_tex.height), - mem_tex.argb.data()); - decoded = decode_texture_bytes(mtra.data(), mtra.size()); - } - } else { - std::vector pack_bytes; - if (read_live_pack_texture(texture_name, pack_bytes)) - decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size()); - } - } -#endif - if (!decoded.ok()) return it->second; - // 40250 CGraphicImageTexture: DXT DDS -> CreateDDSTexture with the file's levels; - // other files -> D3DXCreateTextureFromFileInMemoryEx(D3DX_DEFAULT) = full chain; - // memory images (mem:, CreateTexture(w, h, 1, ...)) -> one level. - const bool is_memory_tex = texture_name.rfind("mem:", 0) == 0; - last_texture_upload_name_ = texture_name; - upload_texture( - decoded.w, decoded.h, decoded.rgba.data(), it->second, true, - !is_memory_tex && decoded.file_mip_levels == 0, &decoded.mips); - it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view); - it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view); - return it->second; - } - - std::uint64_t sampler_state_key(const Render3DDraw& draw, int stage) const { - return (std::uint64_t(draw.address_u[stage] & 15u)) | - (std::uint64_t(draw.address_v[stage] & 15u) << 4) | - (std::uint64_t(draw.min_filter[stage] & 15u) << 8) | - (std::uint64_t(draw.mag_filter[stage] & 15u) << 12) | - (std::uint64_t(draw.mip_filter[stage] & 15u) << 16); - } - - VkSampler sampler_for_draw(const Render3DDraw& draw, int stage) { - const auto key = sampler_state_key(draw, stage); - if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_; - if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second; - auto address_mode = [](std::uint32_t mode) { - switch (mode) { - case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT; - case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER; - default: return VK_SAMPLER_ADDRESS_MODE_REPEAT; - } - }; - VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; - info.addressModeU = address_mode(draw.address_u[stage]); - info.addressModeV = address_mode(draw.address_v[stage]); - info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; - info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE; - info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; - info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; - info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR; - info.minLod = 0.0f; - info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE; - if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) { - info.anisotropyEnable = VK_TRUE; - info.maxAnisotropy = max_anisotropy_; - } - VkSampler sampler = VK_NULL_HANDLE; - check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state"); - state_samplers_.emplace(key, sampler); - return sampler; - } - - VkDescriptorSet get_texture_descriptor( - const std::string& texture0_name, - const std::string& mask_name, - const std::unordered_map>& capture_textures, - const Render3DDraw& draw) { - const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); - const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures); - const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" + - std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1)); - if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { - it->second.last_used_frame = frame_number_; - return it->second.set; - } - const VkDescriptorSet set = allocate_texture_descriptor_set( - tex0.view ? tex0.view : fallback_texture_.view, - tex1.view ? tex1.view : fallback_texture_.view, - sampler_for_draw(draw, 0), sampler_for_draw(draw, 1)); - paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); - return set; - } - - VkDescriptorSet get_ui_texture_descriptor( - const std::string& texture0_name, - const std::string& mask_name, - const std::unordered_map>& capture_textures) { - const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); - if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; - const auto& tex1 = get_gpu_texture(mask_name, capture_textures); - if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; - const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name; - if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { - it->second.last_used_frame = frame_number_; - return it->second.set; - } - const VkDescriptorSet set = allocate_ui_texture_descriptor_set( - tex0.view ? tex0.view : fallback_texture_.view, - tex1.view ? tex1.view : fallback_texture_.view); - paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); - return set; - } - - void upload_texture( - std::uint32_t width, - std::uint32_t height, - const std::uint8_t* rgba, - GpuTexture& texture, - bool count_upload, - bool generate_mips = true, - const std::vector>* file_mips = nullptr) { - const ScopedMs timer(timings_.texture_upload_ms); - const bool has_file_mips = !generate_mips && file_mips && !file_mips->empty(); - VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; - if (has_file_mips) - for (const auto& level : *file_mips) byte_size += level.size(); - const std::uint32_t mip_levels = generate_mips - ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) - : has_file_mips ? 1u + static_cast(file_mips->size()) : 1u; - // Inside a frame the copy is recorded ahead of that frame's render pass and reads the slot's - // staging ring; the GPU runs it with the frame, no queue wait. Outside a frame (the fallback - // texture at startup) it is a one-off submit that waits. - const bool in_frame = f_->recording; - VkBuffer staging_buffer = VK_NULL_HANDLE; - VkDeviceMemory staging_memory = VK_NULL_HANDLE; - VkDeviceSize staging_offset = 0; - std::uint8_t* mapped = nullptr; - const VkDeviceSize aligned_offset = - (f_->staging_used + staging_alignment_ - 1) / staging_alignment_ * staging_alignment_; - if (in_frame && f_->staging_mapped && aligned_offset + byte_size <= f_->staging_capacity) { - staging_buffer = f_->staging_buffer; - staging_offset = aligned_offset; - mapped = f_->staging_mapped + staging_offset; - f_->staging_used = aligned_offset + byte_size; - } else { - if (in_frame) f_->staging_wanted = std::max(f_->staging_wanted, aligned_offset + byte_size); - void* raw = nullptr; - create_host_buffer(byte_size, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, staging_buffer, staging_memory, &raw); - mapped = static_cast(raw); - } - std::memcpy(mapped, rgba, VkDeviceSize(width) * height * 4); - if (has_file_mips) { - auto* dst = mapped + VkDeviceSize(width) * height * 4; - for (const auto& level : *file_mips) { std::memcpy(dst, level.data(), level.size()); dst += level.size(); } - } - if (staging_memory) vkUnmapMemory(device_, staging_memory); - - VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - image_info.imageType = VK_IMAGE_TYPE_2D; - image_info.format = VK_FORMAT_R8G8B8A8_UNORM; - image_info.extent = {width, height, 1}; - image_info.mipLevels = mip_levels; - image_info.arrayLayers = 1; - image_info.samples = VK_SAMPLE_COUNT_1_BIT; - image_info.tiling = VK_IMAGE_TILING_OPTIMAL; - image_info.usage = - VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT; - image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture"); - VkMemoryRequirements image_reqs{}; - vkGetImageMemoryRequirements(device_, texture.image, &image_reqs); - VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - image_alloc.allocationSize = image_reqs.size; - image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); - check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); - - VkCommandBuffer upload_cmd = f_->command; - if (!in_frame) { - VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; - cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; - check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); - VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; - begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; - check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); - } - - VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; - to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_transfer.image = texture.image; - to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &to_transfer); - - std::vector regions(has_file_mips ? mip_levels : 1u); - VkDeviceSize region_offset = 0; - for (std::uint32_t i = 0; i < regions.size(); ++i) { - const std::uint32_t lw = std::max(1u, width >> i), lh = std::max(1u, height >> i); - regions[i].bufferOffset = staging_offset + region_offset; - regions[i].imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; - regions[i].imageExtent = {lw, lh, 1}; - region_offset += VkDeviceSize(lw) * lh * 4; - } - vkCmdCopyBufferToImage( - upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, - static_cast(regions.size()), regions.data()); - - std::int32_t mip_w = static_cast(width); - std::int32_t mip_h = static_cast(height); - for (std::uint32_t i = 1; generate_mips && i < mip_levels; ++i) { - VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - barrier.image = texture.image; - barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1}; - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &barrier); - - const std::int32_t next_w = std::max(1, mip_w / 2); - const std::int32_t next_h = std::max(1, mip_h / 2); - VkImageBlit blit{}; - blit.srcOffsets[0] = {0, 0, 0}; - blit.srcOffsets[1] = {mip_w, mip_h, 1}; - blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1}; - blit.dstOffsets[0] = {0, 0, 0}; - blit.dstOffsets[1] = {next_w, next_h, 1}; - blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; - vkCmdBlitImage( - upload_cmd, - texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, - texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, - 1, &blit, VK_FILTER_LINEAR); - - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &barrier); - - mip_w = next_w; - mip_h = next_h; - } - - VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_shader.image = texture.image; - to_shader.subresourceRange = generate_mips - ? VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1} - : VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &to_shader); - - if (in_frame) { - if (staging_memory) f_->retired_buffers.emplace_back(staging_buffer, staging_memory); - } else { - check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); - VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; - submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; - check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); - check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); - vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); - vkDestroyBuffer(device_, staging_buffer, nullptr); - vkFreeMemory(device_, staging_memory, nullptr); - } - - VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; - view_info.format = VK_FORMAT_R8G8B8A8_UNORM; - view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; - check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture"); - - if (count_upload) { - ++texture_upload_count_; - texture_uploaded_bytes_ += byte_size; - } - } - - void release_texture(GpuTexture& texture) { - for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor}) - if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor) - vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set); - if (texture.view) vkDestroyImageView(device_, texture.view, nullptr); - if (texture.image) vkDestroyImage(device_, texture.image, nullptr); - if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr); - texture = {}; - } - - void prune_textures() { - constexpr std::uint64_t static_retention_frames = 120; - constexpr std::uint64_t memory_retention_frames = 2; - for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) { - const auto retention = it->first.find("mem:") != std::string::npos - ? memory_retention_frames : static_retention_frames; - if (frame_number_ - it->second.last_used_frame > retention) { - vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set); - it = paired_descriptors_.erase(it); - } else ++it; - } - for (auto it = textures_.begin(); it != textures_.end();) { - const auto retention = it->first.rfind("mem:", 0) == 0 - ? memory_retention_frames : static_retention_frames; - if (frame_number_ - it->second.last_used_frame > retention) { - release_texture(it->second); - it = textures_.erase(it); - } else ++it; - } - } - - void release_geometry(Geometry& geometry) { - if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr); - if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr); - geometry = {}; - } - - void upload_geometry(const Render3DDraw& draw, Geometry& geometry) { - const ScopedMs timer(timings_.geometry_upload_ms); - if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) - throw std::runtime_error("invalid native geometry array size"); - const auto count = draw.positions.size() / 3; - if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX) - throw std::runtime_error("geometry exceeds Vulkan index range"); - // D3D8 feeds opaque white for a vertex format without D3DFVF_DIFFUSE. - const std::uint32_t default_argb = 0xffffffffu; - const bool has_skin = - draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4; - std::vector vertices(count); - for (std::size_t i = 0; i < count; ++i) { - vertices[i].position[0] = draw.positions[3*i]; - vertices[i].position[1] = draw.positions[3*i+1]; - vertices[i].position[2] = draw.positions[3*i+2]; - vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f; - vertices[i].vertex_fog = i < draw.vertex_fog.size() ? draw.vertex_fog[i] : 1.0f; - if (3*i + 2 < draw.normals.size()) { - vertices[i].normal[0] = draw.normals[3*i]; - vertices[i].normal[1] = draw.normals[3*i+1]; - vertices[i].normal[2] = draw.normals[3*i+2]; - } else { - vertices[i].normal[0] = 0.0f; - vertices[i].normal[1] = 0.0f; - vertices[i].normal[2] = 1.0f; - } - if (2*i + 1 < draw.uv0.size()) { - vertices[i].uv[0] = draw.uv0[2*i]; - vertices[i].uv[1] = draw.uv0[2*i+1]; - } else { - vertices[i].uv[0] = 0.0f; - vertices[i].uv[1] = 0.0f; - } - const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb; - vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f; - vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f; - vertices[i].color[2] = float(argb & 255) / 255.0f; - vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f; - if (has_skin) { - for (int k = 0; k < 4; ++k) { - vertices[i].joints[k] = draw.bone_indices[4*i + k]; - vertices[i].weights[k] = draw.bone_weights[4*i + k]; - } - } else { - vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0; - vertices[i].weights[0] = 1.0f; - vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f; - } - if (2*i + 1 < draw.uv1.size()) { - vertices[i].mask_uv[0] = draw.uv1[2*i]; - vertices[i].mask_uv[1] = draw.uv1[2*i+1]; - } else { - vertices[i].mask_uv[0] = 0.0f; - vertices[i].mask_uv[1] = 0.0f; - } - } - for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count"); - geometry.index_offset = vertices.size() * sizeof(Vertex); - const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t); - if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB"); - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer"); - VkMemoryRequirements requirements{}; - vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements); - VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - allocation.allocationSize = requirements.size; - allocation.memoryTypeIndex = find_memory_type( - requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory"); - check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory"); - void* mapped = nullptr; - check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory"); - std::memcpy(mapped, vertices.data(), geometry.index_offset); - std::memcpy(static_cast(mapped) + geometry.index_offset, - draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); - vkUnmapMemory(device_, geometry.memory); - geometry.vertex_count = static_cast(count); - geometry.index_count = static_cast(draw.indices.size()); - ++upload_count_; - uploaded_bytes_ += bytes; - } - - bool vsync_ = true; - VkPresentModeKHR present_mode_ = VK_PRESENT_MODE_FIFO_KHR; - float timestamp_period_ns_ = 1.0f; - float max_anisotropy_ = 1.0f; - SDL_Window* window_ = nullptr; - unsigned screen_keyboard_serial_ = 0; - int safe_inset_px_ = 0; - bool show_fps_ = false; - int fps_ = -1; - int fps_frames_ = 0; - std::chrono::steady_clock::time_point fps_window_start_{}; - - // Adds the scope's wall time to a Timings field. - struct ScopedMs { - explicit ScopedMs(double& into) : into_(into), start_(std::chrono::steady_clock::now()) {} - ~ScopedMs() { into_ += std::chrono::duration(std::chrono::steady_clock::now() - start_).count(); } - double& into_; - std::chrono::steady_clock::time_point start_; - }; - - // Presented frames over the last half second. - void update_fps_counter() { - const auto now = std::chrono::steady_clock::now(); - if (fps_frames_++ == 0) { - fps_window_start_ = now; - return; - } - const double elapsed = std::chrono::duration(now - fps_window_start_).count(); - if (elapsed >= 0.5) { - fps_ = int(std::lround((fps_frames_ - 1) / elapsed)); - fps_frames_ = 1; - fps_window_start_ = now; - } - } - VkInstance instance_ = VK_NULL_HANDLE; - VkSurfaceKHR surface_ = VK_NULL_HANDLE; - VkPhysicalDevice physical_ = VK_NULL_HANDLE; - VkDevice device_ = VK_NULL_HANDLE; - VkQueue queue_ = VK_NULL_HANDLE; - std::uint32_t queue_family_ = 0; - std::string device_name_; - VkSwapchainKHR swapchain_ = VK_NULL_HANDLE; - VkFormat swapchain_format_ = VK_FORMAT_UNDEFINED; - VkExtent2D extent_{}; - std::vector image_views_; - std::vector swapchain_images_; - void note_input(const SDL_Event& event) { - switch (event.type) { - case SDL_EVENT_FINGER_DOWN: ++fingers_down_; break; - case SDL_EVENT_FINGER_UP: - case SDL_EVENT_FINGER_CANCELED: fingers_down_ = std::max(0, fingers_down_ - 1); break; - case SDL_EVENT_MOUSE_BUTTON_DOWN: ++buttons_down_; break; - case SDL_EVENT_MOUSE_BUTTON_UP: buttons_down_ = std::max(0, buttons_down_ - 1); break; - case SDL_EVENT_WINDOW_FOCUS_LOST: case SDL_EVENT_WILL_ENTER_BACKGROUND: - // An up event lost with the focus must not keep the client out of the idle rate for good. - fingers_down_ = buttons_down_ = 0; - break; - case SDL_EVENT_FINGER_MOTION: case SDL_EVENT_MOUSE_MOTION: case SDL_EVENT_MOUSE_WHEEL: - case SDL_EVENT_KEY_DOWN: case SDL_EVENT_KEY_UP: case SDL_EVENT_TEXT_INPUT: - case SDL_EVENT_WINDOW_FOCUS_GAINED: case SDL_EVENT_DID_ENTER_FOREGROUND: break; - default: return; - } - last_input_ = std::chrono::steady_clock::now(); - } - std::chrono::steady_clock::time_point last_input_ = std::chrono::steady_clock::now(); - int fingers_down_ = 0, buttons_down_ = 0; - - // --screenshot-out: read back the swapchain image of one frame into a BMP file. - std::string screenshot_path_; - std::uint64_t screenshot_frame_ = 0; - VkImage depth_image_ = VK_NULL_HANDLE; - VkDeviceMemory depth_memory_ = VK_NULL_HANDLE; - VkImageView depth_view_ = VK_NULL_HANDLE; - VkSampleCountFlagBits samples_ = VK_SAMPLE_COUNT_1_BIT; - VkImage msaa_image_ = VK_NULL_HANDLE; - VkDeviceMemory msaa_memory_ = VK_NULL_HANDLE; - VkImageView msaa_view_ = VK_NULL_HANDLE; - VkRenderPass pass_ = VK_NULL_HANDLE; - std::vector framebuffers_; - VkSampler sampler_ = VK_NULL_HANDLE; - VkSampler clamp_sampler_ = VK_NULL_HANDLE; - VkSampler ui_sampler_ = VK_NULL_HANDLE; - std::unordered_map state_samplers_; - VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE; - VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE; - VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE; - // Everything one frame writes while the GPU may still read the previous frame's copy: with - // kFramesInFlight slots the CPU prepares frame N while the GPU draws frame N-1. A slot is reused - // only after its fence (the submit of frame N - frames_in_flight_) has signalled. - struct FrameSlot { - VkCommandBuffer command = VK_NULL_HANDLE; - VkFence fence = VK_NULL_HANDLE; - VkSemaphore acquire = VK_NULL_HANDLE; - std::uint32_t query_base = 0; - bool has_pending_query = false; - bool recording = false; - VkDescriptorSet bone_descriptor_set = VK_NULL_HANDLE; - VkBuffer bone_buffer = VK_NULL_HANDLE; - VkDeviceMemory bone_memory = VK_NULL_HANDLE; - float* bone_mapped = nullptr; - std::size_t bone_capacity = 0; - VkBuffer state_buffer = VK_NULL_HANDLE; - VkDeviceMemory state_memory = VK_NULL_HANDLE; - std::uint8_t* state_mapped = nullptr; - std::size_t state_capacity = 0; - VkBuffer ui_buffer = VK_NULL_HANDLE; - VkDeviceMemory ui_memory = VK_NULL_HANDLE; - std::uint8_t* ui_mapped = nullptr; - std::size_t ui_vertex_capacity = 0, ui_index_capacity = 0; - VkDeviceSize ui_vertex_bytes = 0; - // Texture uploads recorded into this frame's command buffer copy from here. - VkBuffer staging_buffer = VK_NULL_HANDLE; - VkDeviceMemory staging_memory = VK_NULL_HANDLE; - std::uint8_t* staging_mapped = nullptr; - VkDeviceSize staging_capacity = 0, staging_used = 0, staging_wanted = 0; - // Uploads larger than the ring get their own staging buffer, freed when the slot comes back. - std::vector> retired_buffers; - }; - static constexpr std::uint32_t kMaxFramesInFlight = 2; - std::array frames_{}; - FrameSlot* f_ = &frames_[0]; - std::uint32_t frames_in_flight_ = kMaxFramesInFlight; - std::uint32_t frame_slot_ = 0; - // Signalled by a frame's submit, waited by the present of that swapchain image. - std::vector rendered_; - VkDeviceSize state_stride_ = 0; - VkDeviceSize staging_alignment_ = 16; - GpuTexture fallback_texture_{}; - std::unordered_map textures_; - std::unordered_map paired_descriptors_; - std::unordered_map render_targets_; - VkRenderPass offscreen_pass_ = VK_NULL_HANDLE; - VkFormat offscreen_format_ = VK_FORMAT_R5G6B5_UNORM_PACK16; - std::size_t last_offscreen_draw_count_ = 0; - VkShaderModule vertex_module_ = VK_NULL_HANDLE; - VkShaderModule fragment_module_ = VK_NULL_HANDLE; - VkPipelineLayout layout_ = VK_NULL_HANDLE; - std::unordered_map pipelines_; - std::unordered_map geometries_; - std::uint64_t frame_number_ = 0; - std::uint64_t timed_frames_ = 0; - std::size_t upload_count_ = 0, uploaded_bytes_ = 0; - std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0; - std::string last_texture_upload_name_; - VkCommandPool pool_ = VK_NULL_HANDLE; - VkQueryPool query_pool_ = VK_NULL_HANDLE; - bool os_cursor_hidden_ = false; - bool hardware_cursor_enabled_ = false; - std::array game_cursors_{}; - SDL_Cursor* fallback_cursor_ = nullptr; - int current_cursor_shape_ = -1; - int last_mx_ = 0, last_my_ = 0; - std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0; - std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0; - std::size_t last_vertex_count_ = 0, last_index_count_ = 0; - Timings timings_{}; - double last_sync_ms_ = 0; -}; - std::vector make_test_draws(std::size_t count, std::size_t triangles_per_draw) { std::vector draws; draws.reserve(count); @@ -3723,591 +54,6 @@ std::vector make_test_draws(std::size_t count, std::size_t triangl return draws; } -void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms = 0.0, - std::vector frame_ms = {}) { - const auto timing = renderer.timings(); - std::sort(frame_ms.begin(), frame_ms.end()); - const auto percentile = [&](double fraction) { - if (frame_ms.empty()) return 0.0; - const auto index = static_cast(std::ceil(fraction * frame_ms.size())) - 1; - return frame_ms[std::min(index, frame_ms.size() - 1)]; - }; - std::cout << "device=" << renderer.device_name() - << " present_mode=" << renderer.present_mode_name() << " msaa=" << renderer.msaa_samples() - << " frames=" << completed - << " draws=" << renderer.draw_count() - << " skinned_draws=" << renderer.skinned_draw_count() - << " ui_batches=" << renderer.ui_batch_count() - << " ui_quads=" << renderer.ui_quad_count() - << " vertices=" << renderer.vertex_count() - << " indices=" << renderer.index_count() - << " uploads=" << renderer.upload_count() - << " uploaded_bytes=" << renderer.uploaded_bytes() - << " texture_uploads=" << renderer.texture_upload_count() - << " texture_bytes=" << renderer.texture_uploaded_bytes() - << " mean_frame_ms=" << (completed ? wall_ms / completed : 0.0) - << " p95_frame_ms=" << percentile(0.95) - << " p99_frame_ms=" << percentile(0.99) - << " max_frame_ms=" << percentile(1.0) - << " game_update_ms=" << (completed ? update_ms / completed : 0.0) - << " sync_ms=" << (completed ? timing.sync_ms / completed : 0.0) - << " prepare_ms=" << (completed ? timing.prepare_ms / completed : 0.0) - << " steady_prepare_ms=" << (completed > 1 ? timing.steady_prepare_ms / (completed - 1) : 0.0) - << " submit_ms=" << (completed ? timing.submit_ms / completed : 0.0) - << " present_ms=" << (completed ? timing.present_ms / completed : 0.0) - << " gpu_ms=" << (timing.gpu_samples ? timing.gpu_ms / double(timing.gpu_samples) : 0.0) << '\n'; -} - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT -// --py-exec: a test hook run once the auto-login run reaches the GameWindow (or, with --login-screen, -// once the LoginWindow is up). -std::string g_py_exec_after_login; -// --perf-log / MT_PERF_LOG=1: src/host/perf_log.h. -bool g_perf_log = false; -// --fps-cap N: the loop starts a frame at most every 1/N s (0 = as fast as vsync allows). -int g_fps_cap = 0; -// --refresh-rate N: the surface's preferred frame rate (Android ANativeWindow_setFrameRate; MainActivity -// also picks the matching display mode). 0 leaves the system default. -int g_refresh_rate = 0; -// --fps-mode N: the frame-rate setting for this run (a test override: the saved setting stays); -// --idle-fps N: the rate after --idle-after S seconds without input (0 = never lower it). -int g_fps_mode = -1; -int g_idle_fps = 30; -double g_idle_after_s = 15.0; - -// The frame-rate setting (system option dialog, platform/EterBase/FrameRateMode.h) turned into a frame -// rate, a display refresh rate and the ADPF budget; --fps-cap / --refresh-rate pin those for a test run. -// Frames the player does not watch closely cost the same power as the ones they do, so after a while -// without input (and no auto hunt running) the rate drops to --idle-fps until the next touch. The display -// is asked for 60 Hz unless the setting wants more: a 120 Hz panel scanning out a 60 fps game costs power -// for nothing. -class FrameRatePolicy { -public: - struct Decision { - int mode = -1; - int fps = 0; // frames per second the loop aims at - int display_hz = 0; // the refresh rate asked of the display - bool idle = false; - bool operator==(const Decision&) const = default; - }; - - bool pinned() const { return g_fps_cap > 0 || g_refresh_rate > 0; } - - void load() { - path_ = settings_path(); - int mode = MtFrameRate::MODE_STANDARD; - if (std::ifstream in{path_}; in) in >> mode; - if (g_fps_mode >= 0) mode = g_fps_mode; - MtFrameRate::Set(mode); - saved_mode_ = MtFrameRate::Get(); - max_hz_ = int(std::lround(android_perf::max_refresh_rate())); - if (max_hz_ < 30) max_hz_ = 60; - } - - Decision decide(double idle_seconds, bool auto_hunt) { - Decision d; - d.mode = MtFrameRate::Get(); - if (d.mode != saved_mode_) save(saved_mode_ = d.mode); - const int active = d.mode == MtFrameRate::MODE_SAVER ? 30 - : d.mode == MtFrameRate::MODE_HIGH ? max_hz_ : std::min(60, max_hz_); - d.idle = g_idle_fps > 0 && g_idle_fps < active && idle_seconds >= g_idle_after_s && !auto_hunt; - d.fps = d.idle ? g_idle_fps : active; - d.display_hz = d.mode == MtFrameRate::MODE_HIGH && !d.idle ? max_hz_ : std::min(60, max_hz_); - return d; - } - - int max_hz() const { return max_hz_; } - -private: - static std::string settings_path() { - std::string dir; - if (char* pref = SDL_GetPrefPath("metin2port", "client")) { - dir = pref; - SDL_free(pref); - } - return dir + "frame_rate.cfg"; - } - - void save(int mode) { - std::ofstream out{path_, std::ios::trunc}; - out << mode << '\n'; - } - - std::string path_; - int saved_mode_ = -1; - int max_hz_ = 60; -}; - -bool py_exec(const std::string& code) { - std::string err; - if (!PythonBoot::RunLine(code.c_str(), &err)) { - std::cerr << "PythonBoot::RunLine failed: " << err << '\n'; - return false; - } - return true; -} - -bool py_eval_true(const char* expr) { - std::string res, err; - return PythonBoot::Evaluate(expr, &res, &err) && (res == "True" || res == "1"); -} - -#ifdef __ANDROID__ -bool hide_mobile_taskbar_quickslots() { - return py_exec( - "_mobile_taskbar = _stream.curPhaseWindow.interface.wndTaskBar\n" - "for _mobile_quickslot in _mobile_taskbar.quickslot:\n" - " _mobile_quickslot.Hide()\n" - "_mobile_taskbar.GetChild('QuickSlotBoard').Hide()\n"); -} -#endif - -#if defined(__APPLE__) && TARGET_OS_OSX -bool configure_macos_numeric_quickslots() { - return py_exec( - "_mac_game = _stream.curPhaseWindow\n" - "_mac_game.onPressKeyDict[app.DIK_5] = lambda: _mac_game._GameWindow__PressQuickSlot(4)\n" - "_mac_game.onPressKeyDict[app.DIK_6] = lambda: _mac_game._GameWindow__PressQuickSlot(5)\n" - "_mac_game.onPressKeyDict[app.DIK_7] = lambda: _mac_game._GameWindow__PressQuickSlot(6)\n" - "_mac_game.onPressKeyDict[app.DIK_8] = lambda: _mac_game._GameWindow__PressQuickSlot(7)\n" - "for _mac_key in (app.DIK_F1, app.DIK_F2, app.DIK_F3, app.DIK_F4):\n" - " del _mac_game.onPressKeyDict[_mac_key]\n" - "_mac_taskbar = _mac_game.interface.wndTaskBar\n" - "for _mac_slot_number in range(5, 9):\n" - " _mac_taskbar.GetChild('slot_%d' % _mac_slot_number).LoadImage('d:/ymir work/ui/game/taskbar/%d.sub' % _mac_slot_number)\n"); -} -#endif - -bool parse_live_server_spec(const std::string& spec, std::string& host, int& auth_port, int& game_port) { - const auto p1 = spec.find(':'); - if (p1 == std::string::npos) return false; - const auto p2 = spec.find(':', p1 + 1); - if (p2 == std::string::npos) return false; - host = spec.substr(0, p1); - auth_port = std::stoi(spec.substr(p1 + 1, p2 - p1 - 1)); - game_port = std::stoi(spec.substr(p2 + 1)); - return !host.empty() && auth_port > 0 && game_port > 0; -} - -int run_live_client( - VulkanWindow& renderer, - const std::string& client_dir, - int frames, - int fake_mobs, - bool gpu_skinning, - bool native_terrain, - bool login_screen, - bool selection_screen, - const std::string& live_server_spec, - const std::string& capture_out, - const std::string& screenshot_out) { - if (fake_mobs > 0) { - const std::string mobs_str = std::to_string(fake_mobs); - setenv("MT_FAKE_MOB_COUNT", mobs_str.c_str(), 1); - } -#ifdef __ANDROID__ - SetTraceErrorObserver([](const char* line) { - SDL_LogError(SDL_LOG_CATEGORY_APPLICATION, "40250 SYSERR: %s", line); - }); -#endif - char resolved_client_dir[4096] = {}; - const std::string abs_client_dir = realpath(client_dir.c_str(), resolved_client_dir) ? std::string(resolved_client_dir) : client_dir; - setenv("MT_40250_CLIENT", abs_client_dir.c_str(), 1); -#ifdef __ANDROID__ - SDL_Log("live stage: pack init %s", abs_client_dir.c_str()); -#endif - if (chdir(abs_client_dir.c_str()) != 0 || !mtpack40250::initialize(".")) { - throw std::runtime_error("cannot initialize 40250 pack at: " + abs_client_dir); - } - - const bool host_hardware_cursor = !renderer.touch_controller.is_enabled(); - PythonBoot::SetHostHardwareCursorEnabled(host_hardware_cursor); - if (host_hardware_cursor) renderer.enable_game_hardware_cursor(); - - SetNativeTerrainRenderEnabled(native_terrain); - SetGpuSkinningEnabled(gpu_skinning); - // The shadow map is drawn by the GPU (offscreen pass); MT_CPU_SHADOW=1 / --cpu-shadow keeps the - // CPU rasterizer (RecordingDevice::rasterize_shadow) for comparison. - const char* cpu_shadow = std::getenv("MT_CPU_SHADOW"); - SetGpuRenderTargetsEnabled(!(cpu_shadow && *cpu_shadow == '1')); - NativeAudioEngine audio; -#ifdef __ANDROID__ - SDL_Log("live stage: audio initialized"); -#endif - - std::string server_host = "127.0.0.1"; - int auth_port = 0; - int game_port = 0; - const bool use_external_server = !live_server_spec.empty(); - FakeLoginServer server; - if (use_external_server) { - if (!parse_live_server_spec(live_server_spec, server_host, auth_port, game_port)) - throw std::runtime_error("invalid --live-server spec (expected HOST:AUTH_PORT:GAME_PORT): " + live_server_spec); - } else { - if (!server.Start()) throw std::runtime_error("FakeLoginServer failed to start on loopback"); - auth_port = server.AuthPort(); - game_port = server.GamePort(); - } - - { - // HiDPI text: rasterise glyph pages at the swapchain's pixels per UI pixel (before any font - // exists). Rounded up: a fractional density (Android 1760x800 UI on 2376x1080 = 1.35) then - // scales finer glyphs down instead of stretching 1x ones. MT_FONT_OVERSAMPLE=1 keeps the - // 40250 bilevel 1x glyphs. - const float density = float(renderer.width()) / float(std::max(1u, renderer.logical_width())); - int oversample = std::clamp(int(std::ceil(density - 0.05f)), 1, 4); - if (const char* env = std::getenv("MT_FONT_OVERSAMPLE")) oversample = std::max(1, std::atoi(env)); - UISetFontOversample(oversample); - } - - std::string error; - const char* env_stdlib = std::getenv("MT_PYTHON_STDLIB"); - std::string stdlib = (env_stdlib && *env_stdlib) ? env_stdlib : bundled_file_path("python27.zip"); -#ifdef __ANDROID__ - if (!env_stdlib || !*env_stdlib) { - std::size_t zip_size = 0; - void* zip = SDL_LoadFile("assets://python27.zip", &zip_size); - if (!zip) zip = SDL_LoadFile("python27.zip", &zip_size); - if (!zip) throw std::runtime_error("cannot load bundled python27.zip"); - char* pref = SDL_GetPrefPath("mtgodot", "native-render"); - if (!pref) { SDL_free(zip); throw std::runtime_error("cannot locate app data directory"); } - stdlib = std::string(pref) + "python27.zip"; - SDL_free(pref); - std::ofstream output(stdlib, std::ios::binary | std::ios::trunc); - output.write(static_cast(zip), static_cast(zip_size)); - SDL_free(zip); - if (!output) throw std::runtime_error("cannot extract bundled python27.zip"); - } -#endif - if (!PythonBoot::Start(stdlib.c_str(), &error)) - throw std::runtime_error("PythonBoot::Start: " + error); -#ifdef __ANDROID__ - SDL_Log("live stage: PythonBoot started"); -#endif - - SetPlatformServerTime(123456789); - PythonBoot::SetUISafeInset(renderer.ui_safe_inset()); - PythonBoot::SetUISize(static_cast(renderer.logical_width()), static_cast(renderer.logical_height())); - if (!PythonBoot::RunMainScript("", &error) || !PythonBoot::IsAppLooping()) - throw std::runtime_error("PythonBoot::RunMainScript: " + error); -#ifdef __ANDROID__ - SDL_Log("live stage: main script started"); -#endif - - const std::unordered_map> empty_textures; - auto pump_until = [&](double max_seconds, int sleep_ms, const auto& cond) -> bool { - const auto deadline = std::chrono::steady_clock::now() + std::chrono::duration(max_seconds); - while (std::chrono::steady_clock::now() < deadline && renderer.poll(true) && PythonBoot::IsAppLooping()) { - PythonBoot::UIUpdate(); - renderer.sync_game_cursor(); - PythonBoot::UIRender(); - audio.pump(); - unsigned ui_w = renderer.width(), ui_h = renderer.height(); - UIRenderGetSize(&ui_w, &ui_h); - renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); - if (cond()) return true; - if (sleep_ms > 0) - std::this_thread::sleep_for(std::chrono::milliseconds(sleep_ms)); - } - return cond(); - }; - - if (!py_exec("import __main__, app, networkModule, introLogin, introSelect, introLoading, game\n" - "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]")) - throw std::runtime_error("failed to locate networkModule.MainStream"); - - if (!pump_until(20.0, 5, [] { return py_eval_true("isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)"); })) - throw std::runtime_error("timed out waiting for LoginWindow"); -#ifdef __ANDROID__ - SDL_Log("live stage: LoginWindow open"); -#endif - - if (login_screen) { - char conn_cmd[512]; - std::snprintf( - conn_cmd, - sizeof(conn_cmd), - "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" - "_w = _stream.curPhaseWindow\n" - "_w._LoginWindow__OpenServerBoard()", - server_host.c_str(), - game_port, - server_host.c_str(), - auth_port); - if (!py_exec(conn_cmd)) - throw std::runtime_error("failed to configure LoginWindow connection info"); - for (int i = 0; i < 15; ++i) { - PythonBoot::UIUpdate(); - renderer.sync_game_cursor(); - PythonBoot::UIRender(); - } - if (!g_py_exec_after_login.empty() && !py_exec(g_py_exec_after_login)) - throw std::runtime_error("--py-exec failed"); - } else { - char login_cmd[512]; - std::snprintf( - login_cmd, - sizeof(login_cmd), - "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" - "_w = _stream.curPhaseWindow\n" - "_w._LoginWindow__OpenLoginBoard()\n" - "_w.idEditLine.SetText('%s')\n" - "_w.pwdEditLine.SetText('%s')\n" - "_w._LoginWindow__OnClickLoginButton()", - server_host.c_str(), - game_port, - server_host.c_str(), - auth_port, - FakeLoginServer::kLogin, - FakeLoginServer::kPassword); -#ifdef __ANDROID__ - SDL_Log("live stage: submitting fake login"); -#endif - if (!py_exec(login_cmd)) - throw std::runtime_error("failed to submit login credentials"); -#ifdef __ANDROID__ - SDL_Log("live stage: fake login submitted"); -#endif - - if (!pump_until(20.0, 5, [&] { - return (use_external_server || server.Has("game1:login2")) && - py_eval_true("isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)"); - })) - throw std::runtime_error("timed out waiting for SelectCharacterWindow"); -#ifdef __ANDROID__ - SDL_Log("live stage: SelectCharacterWindow open"); -#endif - - if (!selection_screen) { - if (!py_exec("_stream.curPhaseWindow.SelectSlot(0)\n" - "_stream.curPhaseWindow.StartGame()")) - throw std::runtime_error("failed to start game from SelectCharacterWindow"); - - if (!use_external_server) { - if (!pump_until(30.0, 5, [&] { return server.Has("game2:client_version"); })) - throw std::runtime_error("timed out waiting for LoadingWindow (client_version)"); - } - - if (!pump_until(60.0, 0, [&] { - return (use_external_server || server.Has("game2:burst_pong")) && - py_eval_true("isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()"); - })) - throw std::runtime_error("timed out waiting for GameWindow (server error: " + server.Error() + ")"); - -#ifdef __ANDROID__ - if (!hide_mobile_taskbar_quickslots()) - throw std::runtime_error("failed to hide the desktop quick-slot bar on Android"); -#endif -#if defined(__APPLE__) && TARGET_OS_OSX - if (!configure_macos_numeric_quickslots()) - throw std::runtime_error("failed to configure numeric quick slots on macOS"); -#endif - if (!g_py_exec_after_login.empty() && !py_exec(g_py_exec_after_login)) - throw std::runtime_error("--py-exec failed"); - - const int warmup_frames = std::max(10, fake_mobs / 2 + 10); - for (int i = 0; i < warmup_frames; ++i) { - if (!renderer.poll(true) || !PythonBoot::IsAppLooping()) break; - PythonBoot::UIUpdate(); - renderer.sync_game_cursor(); - audio.pump(); - } - } - } - - // Render one warmup frame to upload initial scene & UI textures/geometries before timed benchmark. - { - unsigned ui_w = renderer.width(), ui_h = renderer.height(); - UIRenderGetSize(&ui_w, &ui_h); - renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); - } - - if (!capture_out.empty()) { - std::unordered_map> cap_textures; - auto add_tex = [&](const std::string& name) { - if (name.empty() || cap_textures.count(name)) return; - if (name.rfind("mem:", 0) == 0) { - UIMemoryTexture mem_tex; - if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { - auto mtra = native_draw_capture::encode_raw_argb_as_mtra( - static_cast(mem_tex.width), - static_cast(mem_tex.height), - mem_tex.argb.data()); - if (!mtra.empty()) cap_textures.emplace(name, std::move(mtra)); - } - return; - } - std::vector bytes; - if (read_live_pack_texture(name, bytes)) - cap_textures.emplace(name, std::move(bytes)); - }; - for (const auto& d : Render3DDraws()) { - add_tex(d.texture0); - add_tex(d.texture1); - } - for (const auto& c : UIRenderCommands()) { - if (c.kind == UIRenderCommand::Image) { - add_tex(c.text); - add_tex(c.mask); - } - } - unsigned ui_w = renderer.width(), ui_h = renderer.height(); - UIRenderGetSize(&ui_w, &ui_h); - std::vector cap_ui = UIRenderCommands(); - if (renderer.touch_controller.is_enabled()) { - renderer.touch_controller.update_screen_size(int(ui_w), int(ui_h)); - renderer.touch_controller.append_ui_commands(cap_ui); - } - native_draw_capture::write(capture_out, Render3DDraws(), cap_textures, ui_w, ui_h, cap_ui); - } - - if (!screenshot_out.empty() && frames > 0) { - renderer.request_screenshot(screenshot_out, renderer.frame_number() + static_cast(frames)); - } - - renderer.reset_timings(); - native_perf::PerfLog perf_log; - if (g_perf_log) { - perf_log.open("device=" + renderer.device_name() + " present_mode=" + renderer.present_mode_name() + - " msaa=" + std::to_string(renderer.msaa_samples()) + " size=" + std::to_string(renderer.width()) + - "x" + std::to_string(renderer.height()) + " gpu_skinning=" + std::to_string(gpu_skinning) + - " terrain=" + std::to_string(native_terrain)); - } - if (g_refresh_rate > 0) { - const bool ok = android_perf::request_frame_rate(renderer.sdl_window(), float(g_refresh_rate)); - SDL_Log("frame rate request %d Hz: %s", g_refresh_rate, ok ? "accepted" : "unavailable"); - } - perf_log.set_fps_cap(g_fps_cap); - perf_log.set_refresh_rate_source([] { return android_perf::display_refresh_rate(); }); - perf_log.set_render_rate_source([] { return android_perf::render_rate(); }); - perf_log.set_battery_source([] { return android_perf::battery_celsius(); }); - perf_log.set_power_source([] { - const auto p = android_perf::battery_power(); - return native_perf::PowerSample{p.current_ma, p.voltage_mv, p.power_mw, p.plugged}; - }); - // ADPF: each frame's CPU work (everything but the vsync/fence waits and the cap's sleep) against the - // frame budget. Opened once the script thread exists, i.e. here. - using steady = std::chrono::steady_clock; - // The target is 80% of the frame period: the governor settles the clocks so the reported work just - // meets the target, so a target equal to the period leaves every other frame late (cap 60 on the - // test phone: 54 fps against the period, 60 against 80% of it). - const auto budget_for = [](int fps) { return std::chrono::nanoseconds(800'000'000LL / std::max(1, fps)); }; - FrameRatePolicy policy; - policy.load(); - const auto frame_budget = budget_for( - g_fps_cap > 0 ? g_fps_cap : (g_refresh_rate > 0 ? g_refresh_rate : 60)); - android_perf::PerformanceHint hint; - { - std::vector tids{android_perf::current_thread_id()}; - if (const int script = PythonBoot::ScriptThreadId()) tids.push_back(script); - const bool ok = hint.open(tids, frame_budget.count()); - SDL_Log("ADPF performance hint (%zu threads, %.2f ms): %s", tids.size(), frame_budget.count() / 1e6, - ok ? "on" : "unavailable"); - } - const auto start = steady::now(); - double update_ms = 0.0; - int completed = 0; - std::vector frame_ms; - if (frames > 0) frame_ms.reserve(static_cast(frames)); - auto cap_period = g_fps_cap > 0 ? std::chrono::nanoseconds(1'000'000'000LL / g_fps_cap) - : std::chrono::nanoseconds(0); - auto next_frame_at = steady::now(); - FrameRatePolicy::Decision decision; - auto display_checked_at = steady::time_point{}; - // The loop sleeps to the policy's rate unless the display already runs at it (vsync then paces the - // loop, and a second clock beating against it would drop frames). - const auto apply_policy = [&] { - if (policy.pinned()) return; - const auto now = steady::now(); - const auto next = policy.decide(renderer.input_idle_seconds(), PythonBoot::AutoHuntIsEnabled()); - const bool changed = !(next == decision); - if (changed) { - if (next.display_hz != decision.display_hz) { - android_perf::set_display_refresh_rate(float(next.display_hz)); - android_perf::request_frame_rate(renderer.sdl_window(), float(next.display_hz)); - } - hint.set_target(budget_for(next.fps).count()); - // Efficiency cores and lower clocks, except where the setting asks for the highest rate. - hint.prefer_power_efficiency(next.mode != MtFrameRate::MODE_HIGH || next.idle); - perf_log.set_fps_cap(next.fps); - perf_log.set_frame_policy(next.mode, next.idle); - SDL_Log("frame rate: mode %d -> %d fps, display %d Hz%s", next.mode, next.fps, next.display_hz, - next.idle ? " (idle)" : ""); - decision = next; - } - if (changed || now - display_checked_at > std::chrono::seconds(1)) { - display_checked_at = now; - const double hz = android_perf::display_refresh_rate(); - const bool vsync_paced = hz > 0 && std::fabs(hz - decision.fps) < 2.0; - const auto period = vsync_paced ? std::chrono::nanoseconds(0) - : std::chrono::nanoseconds(1'000'000'000LL / decision.fps); - if (period != cap_period) { - cap_period = period; - next_frame_at = now; - } - } - }; - while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) { - apply_policy(); - double pace_ms = 0.0; - if (cap_period.count()) { - // Fixed cadence: a late frame starts the next one at once, a frame more than one period - // late re-anchors instead of bursting to catch up. - const auto now = steady::now(); - if (next_frame_at > now) { - std::this_thread::sleep_until(next_frame_at); - pace_ms = std::chrono::duration(steady::now() - now).count(); - } else if (now - next_frame_at > cap_period) { - next_frame_at = now; - } - next_frame_at += cap_period; - } - const auto frame_start = steady::now(); - const auto t0 = frame_start; - PythonBoot::UIUpdate(); - const auto t_app = std::chrono::steady_clock::now(); - renderer.sync_game_cursor(); - PythonBoot::UIRender(); - audio.pump(); - const auto t1 = std::chrono::steady_clock::now(); - update_ms += std::chrono::duration(t1 - t0).count(); - unsigned ui_w = renderer.width(), ui_h = renderer.height(); - UIRenderGetSize(&ui_w, &ui_h); - renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); - const auto t2 = std::chrono::steady_clock::now(); - const double elapsed_ms = std::chrono::duration(t2 - frame_start).count(); - PythonBoot::SetNativeRenderFrameTime(static_cast(elapsed_ms)); - hint.report(std::int64_t((elapsed_ms - renderer.last_sync_ms()) * 1e6)); - if (perf_log.enabled()) { - native_perf::FrameInput in; - in.frame_ms = elapsed_ms; - in.pace_ms = pace_ms; - in.app_ms = std::chrono::duration(t_app - t0).count(); - in.host_ms = std::chrono::duration(t1 - t_app).count(); - in.render_ms = std::chrono::duration(t2 - t1).count(); - in.renderer = renderer.perf_totals(); - perf_log.frame(in, [] { return PythonBoot::CurrentMapName(); }); - } - frame_ms.push_back(elapsed_ms); - ++completed; - } - renderer.finish_gpu_timings(); -#ifdef __ANDROID__ - SDL_Log("live stage: render loop complete (%d frames)", completed); -#endif - const auto wall_ms = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); - print_summary(renderer, completed, wall_ms, update_ms, std::move(frame_ms)); - - PythonBoot::Stop(); -#ifdef __ANDROID__ - SDL_Log("live stage: PythonBoot stopped"); -#endif - if (!use_external_server) - server.Stop(); - return (frames == 0 || completed == frames) ? 0 : 2; -} -#endif - } // namespace int main(int argc, char** argv) { diff --git a/src/host/native_audio.cpp b/src/host/native_audio.cpp new file mode 100644 index 00000000..74b04b7b --- /dev/null +++ b/src/host/native_audio.cpp @@ -0,0 +1,267 @@ +#include "native_audio.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "texture_decode.h" +#include "platform/MilesLib/AudioCommands.h" + +#include +#include +#include +#include + +namespace mt_host { + +#ifdef __APPLE__ + +OSStatus NativeAudioEngine::mem_audio_read_proc( + void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) { + const auto* mem = static_cast(inClientData); + if (inPosition < 0 || static_cast(inPosition) >= mem->size) { + *actualCount = 0; + return noErr; + } + const std::size_t avail = mem->size - static_cast(inPosition); + const std::size_t to_read = std::min(requestCount, avail); + std::memcpy(buffer, mem->data + inPosition, to_read); + *actualCount = static_cast(to_read); + return noErr; +} + +SInt64 NativeAudioEngine::mem_audio_get_size_proc(void* inClientData) { + return static_cast(static_cast(inClientData)->size); +} +#endif + +NativeAudioEngine::NativeAudioEngine() { + if (SDL_InitSubSystem(SDL_INIT_AUDIO)) { + SDL_AudioSpec spec{}; + spec.format = SDL_AUDIO_F32; + spec.channels = 2; + spec.freq = 44100; + stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_PLAYBACK, &spec, nullptr, nullptr); + if (stream_) + SDL_ResumeAudioStreamDevice(stream_); + } + // Always DrainAudioCommands() from cold boot so stale queued commands don't accumulate. + (void)DrainAudioCommands(); +} + +NativeAudioEngine::~NativeAudioEngine() { + if (stream_) + SDL_DestroyAudioStream(stream_); + SDL_QuitSubSystem(SDL_INIT_AUDIO); +} + +void NativeAudioEngine::pump() { + const auto commands = DrainAudioCommands(); + for (const auto& cmd : commands) { + ++commands_drained_; + switch (cmd.type) { + case AudioCommand::PlaySound2D: { + const auto* clip = get_clip(cmd.filename); + if (clip && !clip->empty()) { + Voice v{}; + v.samples = clip; + v.volume = sound_volume_; + v.target_volume = sound_volume_; + v.filename = cmd.filename; + voices_.push_back(std::move(v)); + ++played_2d_; + } + break; + } + case AudioCommand::PlaySound3D: { + const auto* clip = get_clip(cmd.filename); + if (clip && !clip->empty()) { + Voice v{}; + v.samples = clip; + v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + v.target_volume = v.volume; + v.loop = cmd.play_count != 1; + v.is_3d = true; + v.handle = cmd.id; + v.filename = cmd.filename; + voices_.push_back(std::move(v)); + ++played_3d_; + } + break; + } + case AudioCommand::StopSound3D: + voices_.erase( + std::remove_if(voices_.begin(), voices_.end(), [&](const Voice& v) { + return v.is_3d && v.handle == cmd.id; + }), + voices_.end()); + break; + case AudioCommand::StopAllSound3D: + voices_.erase( + std::remove_if(voices_.begin(), voices_.end(), [](const Voice& v) { return v.is_3d; }), + voices_.end()); + break; + case AudioCommand::SetSoundVolume3D: + for (auto& v : voices_) { + if (v.is_3d && v.handle == cmd.id) { + v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + v.target_volume = v.volume; + } + } + break; + case AudioCommand::PlayMusic: + case AudioCommand::FadeInMusic: { + const auto* clip = get_clip(cmd.filename); + if (clip && !clip->empty()) { + bgm_.samples = clip; + bgm_.frame_cursor = 0; + bgm_.loop = true; + bgm_.filename = cmd.filename; + bgm_.stop_after_fade = false; + const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + bgm_.target_volume = target; + if (cmd.type == AudioCommand::FadeInMusic) { + bgm_.volume = 0.0f; + bgm_.fade_step_per_frame = target / (44100.0f * 1.5f); + } else { + bgm_.volume = target; + bgm_.fade_step_per_frame = 0.0f; + } + ++played_music_; + } + break; + } + case AudioCommand::FadeOutMusic: + case AudioCommand::FadeOutAllMusic: + if (bgm_.samples) { + bgm_.target_volume = 0.0f; + bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f); + bgm_.stop_after_fade = true; + } + break; + case AudioCommand::FadeLimitOutMusic: + if (bgm_.samples) { + bgm_.target_volume = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + bgm_.fade_step_per_frame = (bgm_.target_volume - bgm_.volume) / (44100.0f * 1.0f); + bgm_.stop_after_fade = false; + } + break; + case AudioCommand::SetMusicVolume: + music_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); + if (bgm_.samples && !bgm_.stop_after_fade) { + bgm_.volume = music_volume_; + bgm_.target_volume = music_volume_; + } + break; + case AudioCommand::SetSoundVolume: + sound_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); + break; + default: + break; + } + } + + if (!stream_) return; + const int queued_bytes = SDL_GetAudioStreamAvailable(stream_); + if (queued_bytes < 0) return; + const std::size_t queued_frames = static_cast(queued_bytes) / (2 * sizeof(float)); + constexpr std::size_t kTargetQueuedFrames = 4410; // ~100 ms stereo @ 44.1 kHz + if (queued_frames >= kTargetQueuedFrames) return; + const std::size_t frames_to_mix = kTargetQueuedFrames - queued_frames; + mix_buffer_.assign(frames_to_mix * 2, 0.0f); + + auto mix_voice = [&](Voice& v) -> bool { + if (!v.samples || v.samples->empty()) return false; + const std::size_t total_frames = v.samples->size() / 2; + if (total_frames == 0) return false; + const float* src = v.samples->data(); + for (std::size_t f = 0; f < frames_to_mix; ++f) { + if (v.frame_cursor >= total_frames) { + if (v.loop) v.frame_cursor = 0; + else return false; + } + if (v.fade_step_per_frame != 0.0f) { + v.volume += v.fade_step_per_frame; + if ((v.fade_step_per_frame > 0.0f && v.volume >= v.target_volume) || + (v.fade_step_per_frame < 0.0f && v.volume <= v.target_volume)) { + v.volume = v.target_volume; + v.fade_step_per_frame = 0.0f; + if (v.stop_after_fade && v.volume <= 0.0001f) { + v.samples = nullptr; + return false; + } + } + } + mix_buffer_[f * 2 + 0] += src[v.frame_cursor * 2 + 0] * v.volume; + mix_buffer_[f * 2 + 1] += src[v.frame_cursor * 2 + 1] * v.volume; + ++v.frame_cursor; + } + return v.loop || v.frame_cursor < total_frames; + }; + + if (bgm_.samples) { + if (!mix_voice(bgm_)) bgm_.samples = nullptr; + } + for (auto it = voices_.begin(); it != voices_.end();) { + if (!mix_voice(*it)) it = voices_.erase(it); + else ++it; + } + for (float& sample : mix_buffer_) + sample = std::clamp(sample, -1.0f, 1.0f); + SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast(mix_buffer_.size() * sizeof(float))); +} + +const std::vector* NativeAudioEngine::get_clip(const std::string& vpath) { + if (vpath.empty()) return nullptr; + auto [it, inserted] = clips_.try_emplace(vpath); + if (!inserted) return &it->second; +#ifdef __APPLE__ + std::vector bytes; + if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16) + return &it->second; + MemoryAudioBuffer mem{bytes.data(), bytes.size()}; + AudioFileID audio_file = nullptr; + if (AudioFileOpenWithCallbacks( + &mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr || + !audio_file) { + return &it->second; + } + ExtAudioFileRef ext_file = nullptr; + if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) { + AudioFileClose(audio_file); + return &it->second; + } + AudioStreamBasicDescription client_format{}; + client_format.mSampleRate = 44100.0; + client_format.mFormatID = kAudioFormatLinearPCM; + client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked; + client_format.mBytesPerPacket = 8; + client_format.mFramesPerPacket = 1; + client_format.mBytesPerFrame = 8; + client_format.mChannelsPerFrame = 2; + client_format.mBitsPerChannel = 32; + if (ExtAudioFileSetProperty( + ext_file, + kExtAudioFileProperty_ClientDataFormat, + sizeof(client_format), + &client_format) == noErr) { + std::vector chunk(4096 * 2); + while (true) { + UInt32 frame_count = 4096; + AudioBufferList buf_list{}; + buf_list.mNumberBuffers = 1; + buf_list.mBuffers[0].mNumberChannels = 2; + buf_list.mBuffers[0].mDataByteSize = static_cast(chunk.size() * sizeof(float)); + buf_list.mBuffers[0].mData = chunk.data(); + if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0) + break; + it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2); + if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes + break; + } + } + ExtAudioFileDispose(ext_file); + AudioFileClose(audio_file); +#endif + return &it->second; +} + +} // namespace mt_host +#endif diff --git a/src/host/native_audio.h b/src/host/native_audio.h new file mode 100644 index 00000000..f886d732 --- /dev/null +++ b/src/host/native_audio.h @@ -0,0 +1,67 @@ +#pragma once + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include +#ifdef __APPLE__ +#include +#endif + +#include +#include +#include +#include +#include + +namespace mt_host { + +// Drains the MilesLib AudioCommand queue and mixes 2D/3D sounds and BGM into an SDL audio stream. +class NativeAudioEngine { + struct MemoryAudioBuffer { + const std::uint8_t* data = nullptr; + std::size_t size = 0; + }; + struct Voice { + const std::vector* samples = nullptr; + std::size_t frame_cursor = 0; + float volume = 1.0f; + float target_volume = 1.0f; + float fade_step_per_frame = 0.0f; + bool stop_after_fade = false; + bool loop = false; + bool is_3d = false; + int handle = 0; + std::string filename; + }; + +#ifdef __APPLE__ + static OSStatus mem_audio_read_proc( + void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount); + + static SInt64 mem_audio_get_size_proc(void* inClientData); +#endif + +public: + NativeAudioEngine(); + + ~NativeAudioEngine(); + + void pump(); + +private: + const std::vector* get_clip(const std::string& vpath); + + SDL_AudioStream* stream_ = nullptr; + std::unordered_map> clips_; + std::vector voices_; + Voice bgm_{}; + std::vector mix_buffer_; + float sound_volume_ = 0.8f; + float music_volume_ = 0.5f; + std::size_t commands_drained_ = 0; + std::size_t played_2d_ = 0; + std::size_t played_3d_ = 0; + std::size_t played_music_ = 0; +}; + +} // namespace mt_host +#endif diff --git a/src/host/render_state.cpp b/src/host/render_state.cpp new file mode 100644 index 00000000..8242c827 --- /dev/null +++ b/src/host/render_state.cpp @@ -0,0 +1,130 @@ +#include "render_state.h" + +namespace mt_host { + +std::array multiply(const float* a, const float* b) { + std::array result{}; + for (int row = 0; row < 4; ++row) + for (int col = 0; col < 4; ++col) + for (int k = 0; k < 4; ++k) + result[row * 4 + col] += a[row * 4 + k] * b[k * 4 + col]; + return result; +} + +std::array draw_mvp(const Render3DDraw& draw) { + const auto world_view = multiply(draw.world, draw.view); + return multiply(world_view.data(), draw.proj); +} + +std::array unpack_argb(std::uint32_t argb) { + return { + float((argb >> 16) & 255) / 255.0f, + float((argb >> 8) & 255) / 255.0f, + float(argb & 255) / 255.0f, + float((argb >> 24) & 255) / 255.0f}; +} + +// Inverse transpose of the upper 3x3 of a row-vector matrix, the D3D8 normal transform. +std::array normal_matrix(const std::array& m) { + const float a = m[0], b = m[1], c = m[2], d = m[4], e = m[5], f = m[6], g = m[8], h = m[9], i = m[10]; + const float A = e * i - f * h, B = -(d * i - f * g), C = d * h - e * g; + const float D = -(b * i - c * h), E = a * i - c * g, F = -(a * h - b * g); + const float G = b * f - c * e, H = -(a * f - c * d), I = a * e - b * d; + const float det = a * A + b * B + c * C; + std::array out = kIdentityMatrix; + if (det == 0.0f) return out; + const float s = 1.0f / det; + // inverse = adjugate / det; its transpose is the cofactor matrix / det. + out[0] = A * s; out[1] = B * s; out[2] = C * s; + out[4] = D * s; out[5] = E * s; out[6] = F * s; + out[8] = G * s; out[9] = H * s; out[10] = I * s; + return out; +} + +PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded) { + PushConstants constants{}; + constants.mvp = draw_mvp(draw); + constants.params = {skin_offset_encoded, draw.pretransformed ? 1.0f : 0.0f, 0.0f, 0.0f}; + return constants; +} + +FixedFunctionState make_fixed_function_state(const Render3DDraw& draw) { + FixedFunctionState state{}; + for (int stage = 0; stage < 2; ++stage) { + state.stage_color[stage] = {draw.color_op[stage], draw.color_arg1[stage], draw.color_arg2[stage], + draw.texcoord_index[stage]}; + state.stage_alpha[stage] = {draw.alpha_op[stage], draw.alpha_arg1[stage], draw.alpha_arg2[stage], + draw.texture_transform_flags[stage]}; + std::copy(draw.texture_matrix[stage], draw.texture_matrix[stage] + 16, state.texture_matrix[stage].begin()); + } + state.texture_factor = unpack_argb(draw.texture_factor); + state.world_view = multiply(draw.world, draw.view); + state.normal_matrix = normal_matrix(state.world_view); + state.fog_color = unpack_argb(draw.fog_color); + state.fog_params = {draw.fog_start, draw.fog_end, draw.fog_density, + draw.fog_range_enable ? 1.0f : 0.0f}; + const std::uint32_t fog_mode = draw.fog_enable + ? (draw.fog_table_mode ? draw.fog_table_mode : (draw.pretransformed ? 4u : draw.fog_vertex_mode)) : 0; + state.flags = {1u, !draw.texture0.empty() ? 1u : 0u, !draw.texture1.empty() ? 1u : 0u, fog_mode}; + state.lighting_flags = {draw.lighting, draw.color_vertex, draw.diffuse.empty() ? 0u : 1u, + draw.normalize_normals}; + state.material_sources = {draw.diffuse_material_source, draw.ambient_material_source, + draw.emissive_material_source, draw.local_viewer}; + state.alpha_test = {draw.alpha_test, draw.alpha_func, draw.alpha_ref, 0}; + std::copy(draw.material_diffuse, draw.material_diffuse + 4, state.material_diffuse.begin()); + std::copy(draw.material_ambient, draw.material_ambient + 4, state.material_ambient.begin()); + std::copy(draw.material_emissive, draw.material_emissive + 4, state.material_emissive.begin()); + state.global_ambient = unpack_argb(draw.ambient); + const float* v = draw.view; + for (int i = 0; i < 8; ++i) { + const auto& light = draw.lights[i]; + const float* p = light.position; + const float* d = light.direction; + std::array position{}, direction{}; + for (int c = 0; c < 3; ++c) { + position[c] = p[0] * v[c] + p[1] * v[4 + c] + p[2] * v[8 + c] + v[12 + c]; + direction[c] = d[0] * v[c] + d[1] * v[4 + c] + d[2] * v[8 + c]; + } + const float length = std::sqrt(direction[0] * direction[0] + direction[1] * direction[1] + + direction[2] * direction[2]); + if (length > 0.0f) + for (float& c : direction) c /= length; + state.light_position_type[i] = {position[0], position[1], position[2], float(light.type)}; + state.light_direction_range[i] = {direction[0], direction[1], direction[2], light.range}; + std::copy(light.diffuse, light.diffuse + 4, state.light_diffuse[i].begin()); + std::copy(light.ambient, light.ambient + 4, state.light_ambient[i].begin()); + state.light_attenuation[i] = {light.attenuation[0], light.attenuation[1], light.attenuation[2], light.falloff}; + state.light_spot[i] = {std::cos(light.theta * 0.5f), std::cos(light.phi * 0.5f), 0, 0}; + } + return state; +} + +std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size) { + const auto* data = static_cast(bytes); + std::size_t i = 0; + for (; i + sizeof(std::uint64_t) <= size; i += sizeof(std::uint64_t)) { + std::uint64_t word = 0; + std::memcpy(&word, data + i, sizeof(word)); + hash = (hash ^ word) * 1099511628211ull; + } + for (; i < size; ++i) hash = (hash ^ data[i]) * 1099511628211ull; + hash = (hash ^ size) * 1099511628211ull; + return hash; +} + +// Immediate-mode draws have no source-buffer key. For them, verify content before reusing a slot. +std::uint64_t geometry_hash(const Render3DDraw& draw) { + std::uint64_t hash = 14695981039346656037ull; + const std::uint32_t flags = (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u); + hash = hash_bytes(hash, &flags, sizeof(flags)); + hash = hash_bytes(hash, draw.positions.data(), draw.positions.size() * sizeof(float)); + hash = hash_bytes(hash, draw.rhw.data(), draw.rhw.size() * sizeof(float)); + hash = hash_bytes(hash, draw.vertex_fog.data(), draw.vertex_fog.size() * sizeof(float)); + hash = hash_bytes(hash, draw.normals.data(), draw.normals.size() * sizeof(float)); + hash = hash_bytes(hash, draw.uv0.data(), draw.uv0.size() * sizeof(float)); + hash = hash_bytes(hash, draw.uv1.data(), draw.uv1.size() * sizeof(float)); + hash = hash_bytes(hash, draw.diffuse.data(), draw.diffuse.size() * sizeof(std::uint32_t)); + return hash_bytes(hash, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); +} + +} // namespace mt_host diff --git a/src/host/render_state.h b/src/host/render_state.h new file mode 100644 index 00000000..51a585a0 --- /dev/null +++ b/src/host/render_state.h @@ -0,0 +1,75 @@ +#pragma once + +#include "RenderCommands3D.h" + +#include +#include +#include + +namespace mt_host { + +struct Vertex { + float position[3]; + float normal[3]; + float uv[2]; + float color[4]; + std::uint8_t joints[4]; + float weights[4]; + float mask_uv[2]; + float rhw = 1.0f; + float vertex_fog = 1.0f; +}; + +struct PushConstants { + std::array mvp{}; + // x: bone palette base + 1 (0 = not skinned), y: D3DFVF_XYZRHW, z: UI mask batch, w: unused. + std::array params{}; +}; +static_assert(sizeof(PushConstants) <= 128, "PushConstants must fit within Vulkan's 128-byte minimum guarantee"); + +// std430 payload selected with a dynamic storage-buffer offset for every draw: the D3D8 +// fixed-function state in effect for the draw (texture stages, texture coordinate processing, +// lighting, material, fog, alpha test). A zero flags.x marks a 2D UI batch. +struct alignas(16) FixedFunctionState { + std::array, 2> stage_color{}; // op, arg1, arg2, D3DTSS_TEXCOORDINDEX + std::array, 2> stage_alpha{}; // op, arg1, arg2, D3DTSS_TEXTURETRANSFORMFLAGS + std::array fog_color{}; + std::array fog_params{}; // start, end, density, range enabled + std::array texture_factor{}; + std::array world_view{}; + std::array flags{}; // fixed function, texture0, texture1, fog mode + std::array normal_matrix{}; // inverse transpose of world * view + std::array lighting_flags{}; // LIGHTING, COLORVERTEX, vertex has diffuse, NORMALIZENORMALS + std::array material_sources{}; // DIFFUSE, AMBIENT, EMISSIVE source, LOCALVIEWER + std::array alpha_test{}; // ALPHATESTENABLE, ALPHAFUNC, ALPHAREF, unused + std::array material_diffuse{}; + std::array material_ambient{}; + std::array material_emissive{}; + std::array global_ambient{}; + std::array, 2> texture_matrix{}; + // Lights in camera space. position.w = D3DLIGHTTYPE (0 = disabled). + std::array, 8> light_position_type{}; + std::array, 8> light_direction_range{}; + std::array, 8> light_diffuse{}; + std::array, 8> light_ambient{}; + std::array, 8> light_attenuation{}; // a0, a1, a2, falloff + std::array, 8> light_spot{}; // cos(theta / 2), cos(phi / 2) +}; +static_assert(sizeof(FixedFunctionState) == 1264); + +inline constexpr std::array kIdentityMatrix = { + 1.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 1.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 1.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 1.0f}; + +std::array multiply(const float* a, const float* b); +std::array draw_mvp(const Render3DDraw& draw); +std::array unpack_argb(std::uint32_t argb); +std::array normal_matrix(const std::array& m); +PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded); +FixedFunctionState make_fixed_function_state(const Render3DDraw& draw); +std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size); +std::uint64_t geometry_hash(const Render3DDraw& draw); + +} // namespace mt_host diff --git a/src/host/shaders/native.vert b/src/host/shaders/native.vert index c5e2cbf4..8ee4186f 100644 --- a/src/host/shaders/native.vert +++ b/src/host/shaders/native.vert @@ -23,7 +23,7 @@ layout(push_constant) uniform DrawConstants { vec4 params; // bone base + 1 (0 = none), pretransformed, UI mask, unused } draw; -// D3D8 fixed-function vertex state; see FixedFunctionState in main.cpp. Lights are in camera space. +// D3D8 fixed-function vertex state; see FixedFunctionState in render_state.h. Lights are in camera space. layout(set = 1, binding = 1, std430) readonly buffer FixedFunctionState { uvec4 stage_color[2]; // op, arg1, arg2, D3DTSS_TEXCOORDINDEX uvec4 stage_alpha[2]; // op, arg1, arg2, D3DTSS_TEXTURETRANSFORMFLAGS diff --git a/src/host/texture_decode.cpp b/src/host/texture_decode.cpp new file mode 100644 index 00000000..f509d077 --- /dev/null +++ b/src/host/texture_decode.cpp @@ -0,0 +1,154 @@ +#include "texture_decode.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "platform/PackBackend.h" +#endif + +#include "stb_image.h" + +#include +#include +#include + +namespace mt_host { + +mtimage::Image decode_tga(const std::uint8_t* data, std::size_t size) { + mtimage::Image out; + if (size < 18) return out; + const std::uint8_t id_len = data[0]; + const std::uint8_t cmap_type = data[1]; + const std::uint8_t img_type = data[2]; + const std::uint32_t w = std::uint32_t(data[12]) | (std::uint32_t(data[13]) << 8); + const std::uint32_t h = std::uint32_t(data[14]) | (std::uint32_t(data[15]) << 8); + const std::uint8_t bpp = data[16]; + const std::uint8_t desc = data[17]; + if (cmap_type != 0 || !w || !h || w > 4096 || h > 4096) return out; + if (img_type != 2 && img_type != 3 && img_type != 10) return out; + const std::size_t bytes_per_pixel = bpp / 8; + if (bytes_per_pixel != 1 && bytes_per_pixel != 3 && bytes_per_pixel != 4) return out; + std::size_t offset = 18 + std::size_t(id_len); + if (offset > size) return out; + + const std::size_t pixel_count = std::size_t(w) * std::size_t(h); + std::vector temp(pixel_count * 4); + auto write_pixel = [&](std::size_t idx, const std::uint8_t* src) { + std::uint8_t* dst = &temp[idx * 4]; + if (bytes_per_pixel == 1) { + dst[0] = dst[1] = dst[2] = src[0]; + dst[3] = 255; + } else if (bytes_per_pixel == 3) { + dst[0] = src[2]; + dst[1] = src[1]; + dst[2] = src[0]; + dst[3] = 255; + } else { + dst[0] = src[2]; + dst[1] = src[1]; + dst[2] = src[0]; + dst[3] = src[3]; + } + }; + + if (img_type == 2 || img_type == 3) { + if (offset + pixel_count * bytes_per_pixel > size) return out; + for (std::size_t i = 0; i < pixel_count; ++i) + write_pixel(i, data + offset + i * bytes_per_pixel); + } else if (img_type == 10) { + std::size_t i = 0; + while (i < pixel_count && offset < size) { + const std::uint8_t header = data[offset++]; + const std::size_t run = (header & 0x7fu) + 1u; + if (header & 0x80u) { + if (offset + bytes_per_pixel > size) return out; + for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) + write_pixel(i, data + offset); + offset += bytes_per_pixel; + } else { + if (offset + run * bytes_per_pixel > size) return out; + for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) { + write_pixel(i, data + offset); + offset += bytes_per_pixel; + } + } + } + if (i != pixel_count) return out; + } + + out.w = static_cast(w); + out.h = static_cast(h); + const bool top_origin = (desc & 0x20u) != 0; + if (top_origin) { + out.rgba = std::move(temp); + } else { + out.rgba.resize(pixel_count * 4); + const std::size_t row_bytes = std::size_t(w) * 4; + for (std::uint32_t y = 0; y < h; ++y) + std::memcpy(&out.rgba[std::size_t(y) * row_bytes], &temp[std::size_t(h - 1 - y) * row_bytes], row_bytes); + } + return out; +} + +mtimage::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size) { + if (!data || size < 12) return {}; + if (data[0] == 'M' && data[1] == 'T' && data[2] == 'R' && data[3] == 'A') { + std::uint32_t w = 0, h = 0; + std::memcpy(&w, data + 4, 4); + std::memcpy(&h, data + 8, 4); + const std::size_t bytes = std::size_t(w) * std::size_t(h) * 4; + if (w > 0 && h > 0 && w <= 4096 && h <= 4096 && size == 12 + bytes) { + mtimage::Image out; + out.w = static_cast(w); + out.h = static_cast(h); + out.rgba.assign(data + 12, data + 12 + bytes); + return out; + } + return {}; + } + if (data[0] == 'D' && data[1] == 'D' && data[2] == 'S' && data[3] == ' ') + return mtimage::load_dds(data, size); + auto tga = decode_tga(data, size); + if (tga.ok()) return tga; + if (size > static_cast(INT_MAX)) return {}; + int w = 0, h = 0, channels = 0; + stbi_uc* pixels = stbi_load_from_memory(data, static_cast(size), &w, &h, &channels, 4); + if (pixels && w > 0 && h > 0 && w <= 4096 && h <= 4096) { + mtimage::Image out; + out.w = static_cast(w); + out.h = static_cast(h); + out.rgba.assign(pixels, pixels + std::size_t(w) * std::size_t(h) * 4); + stbi_image_free(pixels); + return out; + } + stbi_image_free(pixels); + return {}; +} + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +bool read_live_pack_texture(const std::string& vpath, std::vector& bytes) { + if (vpath.empty() || !mtpack40250::ready()) + return false; + std::string norm = vpath; + for (char& ch : norm) + if (ch == '\\') ch = '/'; + std::string stripped = norm; + if (stripped.size() >= 2 && stripped[1] == ':') + stripped = stripped.substr(2); + while (!stripped.empty() && stripped.front() == '/') + stripped.erase(stripped.begin()); + auto lower = [](std::string s) { + for (char& ch : s) + if (ch >= 'A' && ch <= 'Z') + ch = static_cast(ch - 'A' + 'a'); + return s; + }; + for (const std::string& candidate : { + norm, stripped, "d:/" + stripped, + lower(norm), lower(stripped), lower("d:/" + stripped)}) { + if (mtpack40250::read(candidate, bytes) && !bytes.empty()) + return true; + } + return false; +} +#endif + +} // namespace mt_host diff --git a/src/host/texture_decode.h b/src/host/texture_decode.h new file mode 100644 index 00000000..1f5cc984 --- /dev/null +++ b/src/host/texture_decode.h @@ -0,0 +1,18 @@ +#pragma once + +#include "dxt.h" + +#include +#include +#include +#include + +namespace mt_host { + +mtimage::Image decode_tga(const std::uint8_t* data, std::size_t size); +mtimage::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +bool read_live_pack_texture(const std::string& vpath, std::vector& bytes); +#endif + +} // namespace mt_host diff --git a/src/host/vulkan_window.cpp b/src/host/vulkan_window.cpp new file mode 100644 index 00000000..03a97a76 --- /dev/null +++ b/src/host/vulkan_window.cpp @@ -0,0 +1,2679 @@ +#include "vulkan_window.h" + +#include "draw_capture.h" +#include "dxt.h" +#include "host_util.h" +#include "input_keymap.h" +#include "texture_decode.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "platform/ScriptLib/PythonBoot.h" +#endif + +#include +#ifdef __APPLE__ +#include +#endif + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace mt_host { + +namespace { + +std::vector read_spirv(const char* name) { + std::size_t size = 0; + void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size); +#ifdef __ANDROID__ + if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size); + if (!bytes) bytes = SDL_LoadFile(name, &size); +#endif + if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name); + if (size == 0 || size % 4) { + SDL_free(bytes); + throw std::runtime_error(std::string("invalid SPIR-V size: ") + name); + } + std::vector words(size / 4); + std::memcpy(words.data(), bytes, size); + SDL_free(bytes); + return words; +} + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +SDL_Cursor* load_game_cursor(int shape) { + static constexpr std::array images = { + "cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub", + "cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub", + "cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub", + "cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"}; + if (shape < 0 || shape >= static_cast(images.size())) return nullptr; + const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape]; + std::vector sub_bytes; + if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr; + std::unordered_map tokens; + std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end())); + std::string line; + while (std::getline(input, line)) { + std::istringstream fields(line); + std::string key, value; + if (!(fields >> key >> value)) continue; + if (!value.empty() && value.front() == '"') { + value.erase(0, 1); + if (!value.empty() && value.back() == '"') value.pop_back(); + } + std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); }); + tokens[key] = value; + } + if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr; + const std::string image_path = tokens["version"] == "2.0" + ? std::string("d:/ymir work/ui/cursor/") + tokens["image"] + : std::string("d:/ymir work/ui/") + tokens["image"]; + std::vector image_bytes; + if (!read_live_pack_texture(image_path, image_bytes)) return nullptr; + const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size()); + if (!image.ok()) return nullptr; + const int left = std::atoi(tokens["left"].c_str()); + const int top = std::atoi(tokens["top"].c_str()); + const int right = std::atoi(tokens["right"].c_str()); + const int bottom = std::atoi(tokens["bottom"].c_str()); + if (left < 0 || top < 0 || right <= left || bottom <= top || + right > image.w || bottom > image.h) return nullptr; + const int width = right - left, height = bottom - top; + std::vector pixels(std::size_t(width) * height * 4); + for (int y = 0; y < height; ++y) + std::memcpy(pixels.data() + std::size_t(y) * width * 4, + image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4, + std::size_t(width) * 4); + SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32, + pixels.data(), width * 4); + if (!surface) return nullptr; + const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0; + const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0; + SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y); + SDL_DestroySurface(surface); + return cursor; +} +#endif + +} // namespace + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + +void VulkanWindow::enable_game_hardware_cursor() { + hardware_cursor_enabled_ = true; + for (int shape = 0; shape < static_cast(game_cursors_.size()); ++shape) { + game_cursors_[shape] = load_game_cursor(shape); + } + fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT); + SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_); + SDL_ShowCursor(); + os_cursor_hidden_ = false; +} + +void VulkanWindow::sync_game_cursor() { + if (!hardware_cursor_enabled_) return; + const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() && + !PythonBoot::IsSoftwareCursorVisible(); + if (visible == os_cursor_hidden_) { + if (visible) SDL_ShowCursor(); + else SDL_HideCursor(); + os_cursor_hidden_ = !visible; + } + if (!visible) return; + const int shape = PythonBoot::CursorShape(); + if (shape == current_cursor_shape_) return; + SDL_Cursor* cursor = shape >= 0 && shape < static_cast(game_cursors_.size()) + ? game_cursors_[shape] : nullptr; + if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_); + current_cursor_shape_ = shape; +} +#endif + +VulkanWindow::VulkanWindow(bool vsync, int init_width, int init_height) : vsync_(vsync) { +#ifdef __APPLE__ + if (!std::getenv("VK_ICD_FILENAMES")) { + for (const char* icd : { + "/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json", + "/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json"}) { + if (std::ifstream(icd).good()) { + setenv("VK_ICD_FILENAMES", icd, 0); + break; + } + } + } +#endif + // Android's back key reaches the game as SDL_SCANCODE_AC_BACK (-> ESC) instead of finishing the activity. + SDL_SetHint(SDL_HINT_ANDROID_TRAP_BACK_BUTTON, "1"); + if (!SDL_Init(SDL_INIT_VIDEO)) throw std::runtime_error(SDL_GetError()); + SDL_SetHint(SDL_HINT_ORIENTATIONS, "LandscapeLeft LandscapeRight"); + window_ = SDL_CreateWindow( + "Metin2 Native Vulkan Client", + std::max(init_width, 320), + std::max(init_height, 240), + SDL_WINDOW_VULKAN | SDL_WINDOW_RESIZABLE | SDL_WINDOW_HIGH_PIXEL_DENSITY); + if (!window_) throw std::runtime_error(SDL_GetError()); + SDL_StartTextInput(window_); + Uint32 extension_count = 0; + const char* const* sdl_extensions = SDL_Vulkan_GetInstanceExtensions(&extension_count); + if (!sdl_extensions) throw std::runtime_error(SDL_GetError()); + std::vector extensions(sdl_extensions, sdl_extensions + extension_count); + std::uint32_t available_count = 0; + check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, nullptr), "vkEnumerateInstanceExtensionProperties count"); + std::vector available(available_count); + check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, available.data()), "vkEnumerateInstanceExtensionProperties"); + const bool portability = std::any_of(available.begin(), available.end(), [](const auto& extension) { + return std::strcmp(extension.extensionName, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; + }); + if (portability && std::find_if(extensions.begin(), extensions.end(), [](const char* name) { + return std::strcmp(name, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; + }) == extensions.end()) + extensions.push_back(VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME); + VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO}; + app.pApplicationName = "Metin2 native renderer"; + app.apiVersion = VK_API_VERSION_1_1; + VkInstanceCreateInfo instance_info{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO}; + instance_info.flags = portability ? VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR : 0; + instance_info.pApplicationInfo = &app; + instance_info.enabledExtensionCount = static_cast(extensions.size()); + instance_info.ppEnabledExtensionNames = extensions.data(); + check(vkCreateInstance(&instance_info, nullptr, &instance_), "vkCreateInstance"); + if (!SDL_Vulkan_CreateSurface(window_, instance_, nullptr, &surface_)) throw std::runtime_error(SDL_GetError()); + select_device(); + VkCommandPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; + pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; + pool_info.queueFamilyIndex = queue_family_; + check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool"); + VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO}; + query_info.queryType = VK_QUERY_TYPE_TIMESTAMP; + query_info.queryCount = 2 * kMaxFramesInFlight; + check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool"); + VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; + for (std::uint32_t i = 0; i < kMaxFramesInFlight; ++i) { + auto& slot = frames_[i]; + VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &alloc, &slot.command), "vkAllocateCommandBuffers"); + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &slot.acquire), "vkCreateSemaphore acquire"); + check(vkCreateFence(device_, &fence_info, nullptr, &slot.fence), "vkCreateFence"); + slot.query_base = 2 * i; + } + create_swapchain(); + create_descriptors_and_buffers(); + create_render_pass_and_layout(); + touch_controller.update_screen_size(int(width()), int(height())); +} + +VulkanWindow::~VulkanWindow() { + if (hardware_cursor_enabled_) { + SDL_SetCursor(nullptr); + for (SDL_Cursor* cursor : game_cursors_) if (cursor) SDL_DestroyCursor(cursor); + if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_); + } + if (device_) vkDeviceWaitIdle(device_); + for (auto& slot : frames_) { + release_retired_buffers(slot); + if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); + if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); + if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); + if (slot.bone_mapped) vkUnmapMemory(device_, slot.bone_memory); + if (slot.bone_buffer) vkDestroyBuffer(device_, slot.bone_buffer, nullptr); + if (slot.bone_memory) vkFreeMemory(device_, slot.bone_memory, nullptr); + if (slot.state_mapped) vkUnmapMemory(device_, slot.state_memory); + if (slot.state_buffer) vkDestroyBuffer(device_, slot.state_buffer, nullptr); + if (slot.state_memory) vkFreeMemory(device_, slot.state_memory, nullptr); + if (slot.ui_mapped) vkUnmapMemory(device_, slot.ui_memory); + if (slot.ui_buffer) vkDestroyBuffer(device_, slot.ui_buffer, nullptr); + if (slot.ui_memory) vkFreeMemory(device_, slot.ui_memory, nullptr); + if (slot.fence) vkDestroyFence(device_, slot.fence, nullptr); + if (slot.acquire) vkDestroySemaphore(device_, slot.acquire, nullptr); + } + if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr); + for (VkSemaphore semaphore : rendered_) vkDestroySemaphore(device_, semaphore, nullptr); + for (auto& entry : geometries_) release_geometry(entry.second); + for (auto& entry : textures_) release_texture(entry.second); + for (auto& entry : render_targets_) release_render_target(entry.second); + if (offscreen_pass_) vkDestroyRenderPass(device_, offscreen_pass_, nullptr); + release_texture(fallback_texture_); + for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); + if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr); + if (fragment_module_) vkDestroyShaderModule(device_, fragment_module_, nullptr); + if (layout_) vkDestroyPipelineLayout(device_, layout_, nullptr); + if (descriptor_pool_) vkDestroyDescriptorPool(device_, descriptor_pool_, nullptr); + if (descriptor_layout_) vkDestroyDescriptorSetLayout(device_, descriptor_layout_, nullptr); + if (bone_descriptor_layout_) vkDestroyDescriptorSetLayout(device_, bone_descriptor_layout_, nullptr); + if (sampler_) vkDestroySampler(device_, sampler_, nullptr); + if (clamp_sampler_) vkDestroySampler(device_, clamp_sampler_, nullptr); + if (ui_sampler_) vkDestroySampler(device_, ui_sampler_, nullptr); + for (const auto& entry : state_samplers_) vkDestroySampler(device_, entry.second, nullptr); + for (auto framebuffer : framebuffers_) vkDestroyFramebuffer(device_, framebuffer, nullptr); + if (pass_) vkDestroyRenderPass(device_, pass_, nullptr); + if (depth_view_) vkDestroyImageView(device_, depth_view_, nullptr); + if (depth_image_) vkDestroyImage(device_, depth_image_, nullptr); + if (depth_memory_) vkFreeMemory(device_, depth_memory_, nullptr); + destroy_msaa_color(); + for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); + if (swapchain_) vkDestroySwapchainKHR(device_, swapchain_, nullptr); + if (pool_) vkDestroyCommandPool(device_, pool_, nullptr); + if (device_) vkDestroyDevice(device_, nullptr); + if (surface_) vkDestroySurfaceKHR(instance_, surface_, nullptr); + if (instance_) vkDestroyInstance(instance_, nullptr); + if (window_) { + SDL_StopTextInput(window_); + SDL_DestroyWindow(window_); + } + SDL_Quit(); +} + +void VulkanWindow::write_bmp(const std::string& path, const std::uint8_t* pixels) const { + const std::uint32_t w = extent_.width, h = extent_.height; + const bool rgba = swapchain_format_ == VK_FORMAT_R8G8B8A8_UNORM || swapchain_format_ == VK_FORMAT_R8G8B8A8_SRGB; + std::vector file(54 + std::size_t(w) * h * 4); + auto put32 = [&](std::size_t at, std::uint32_t value) { std::memcpy(&file[at], &value, 4); }; + file[0] = 'B'; file[1] = 'M'; + put32(2, std::uint32_t(file.size())); put32(10, 54); put32(14, 40); + put32(18, w); put32(22, h); put32(26, 1u | (32u << 16)); put32(34, w * h * 4); + for (std::uint32_t y = 0; y < h; ++y) + for (std::uint32_t x = 0; x < w; ++x) { + const std::uint8_t* src = pixels + (std::size_t(y) * w + x) * 4; + std::uint8_t* dst = &file[54 + (std::size_t(h - 1 - y) * w + x) * 4]; + dst[0] = rgba ? src[2] : src[0]; dst[1] = src[1]; dst[2] = rgba ? src[0] : src[2]; dst[3] = 255; + } + std::ofstream out(path, std::ios::binary); + out.write(reinterpret_cast(file.data()), std::streamsize(file.size())); +} + +double VulkanWindow::input_idle_seconds() const { + if (fingers_down_ > 0 || buttons_down_ > 0) return 0.0; + return std::chrono::duration(std::chrono::steady_clock::now() - last_input_).count(); +} + +void VulkanWindow::request_screenshot(std::string path, std::uint64_t frame) { + screenshot_path_ = std::move(path); + screenshot_frame_ = frame; +} + +std::uint32_t VulkanWindow::logical_width() const { + int w = 0, h = 0; + if (window_) SDL_GetWindowSize(window_, &w, &h); +#ifdef __ANDROID__ + // Keep the 40250 UI at the same height as the macOS 1280x800 window, + // while retaining the device aspect ratio on wider phone displays. + if (w > 0 && h > 0) return std::max(1, int(std::lround(double(w) * 800.0 / h))); +#endif + return w > 0 ? static_cast(w) : extent_.width; +} + +std::uint32_t VulkanWindow::logical_height() const { + int w = 0, h = 0; + if (window_) SDL_GetWindowSize(window_, &w, &h); +#ifdef __ANDROID__ + if (w > 0 && h > 0) return 800; +#endif + return h > 0 ? static_cast(h) : extent_.height; +} + +VkViewport VulkanWindow::full_viewport() const { + return {0, 0, float(extent_.width), float(extent_.height), 0, 1}; +} + +int VulkanWindow::ui_safe_inset() const { + int pw = 0, ph = 0; + if (safe_inset_px_ <= 0 || !window_ || !SDL_GetWindowSizeInPixels(window_, &pw, &ph) || ph <= 0) return 0; + return int(std::lround(double(safe_inset_px_) * logical_height() / ph)); +} + +VkViewport VulkanWindow::draw_viewport(const Render3DDraw& draw) const { + if (draw.viewport[2] <= 0 || draw.viewport[3] <= 0) return full_viewport(); + const float sx = float(extent_.width) / float(logical_width()); + const float sy = float(extent_.height) / float(logical_height()); + return {(draw.viewport[0] + draw.ui_offset_x) * sx, draw.viewport[1] * sy, draw.viewport[2] * sx, draw.viewport[3] * sy, + draw.viewport_z[0], draw.viewport_z[1]}; +} + +void VulkanWindow::sync_screen_keyboard() { + const bool want = touch_controller.wants_screen_keyboard(); + const unsigned serial = touch_controller.screen_keyboard_serial(); + if (want && (!SDL_TextInputActive(window_) || serial != screen_keyboard_serial_)) { + SDL_PropertiesID props = SDL_CreateProperties(); + SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_TYPE_NUMBER, SDL_TEXTINPUT_TYPE_TEXT); + SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_CAPITALIZATION_NUMBER, SDL_CAPITALIZE_NONE); + SDL_SetBooleanProperty(props, SDL_PROP_TEXTINPUT_AUTOCORRECT_BOOLEAN, false); + SDL_StartTextInputWithProperties(window_, props); + SDL_DestroyProperties(props); + } else if (!want && SDL_TextInputActive(window_)) { + SDL_StopTextInput(window_); + } + screen_keyboard_serial_ = serial; +} + +bool VulkanWindow::poll(bool forward_to_live_client) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + auto map_mouse = [&](float wx, float wy, int& ox, int& oy) { + int win_w = 0, win_h = 0; + SDL_GetWindowSize(window_, &win_w, &win_h); + unsigned ui_w = extent_.width ? extent_.width : 960; + unsigned ui_h = extent_.height ? extent_.height : 640; + UIRenderGetSize(&ui_w, &ui_h); + if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) { + ox = static_cast(std::lround(wx * float(ui_w) / float(win_w))); + oy = static_cast(std::lround(wy * float(ui_h) / float(win_h))); + } else { + ox = static_cast(std::lround(wx)); + oy = static_cast(std::lround(wy)); + } + }; + std::optional> pending_mouse_move; + auto flush_mouse_move = [&] { + if (!pending_mouse_move) return; + const auto [mx, my] = *pending_mouse_move; + pending_mouse_move.reset(); + last_mx_ = mx; + last_my_ = my; + if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return; + PythonBoot::UIMouseMove(mx, my); + }; +#endif + SDL_Event event; + while (SDL_PollEvent(&event)) { + note_input(event); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + // Touches arrive as SDL_EVENT_FINGER_*; the touch-synthesized mouse copy + // would register as a second finger (101) and turn every drag into a pinch. + if (touch_controller.is_enabled() && + ((event.type == SDL_EVENT_MOUSE_MOTION && event.motion.which == SDL_TOUCH_MOUSEID) || + ((event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) && + event.button.which == SDL_TOUCH_MOUSEID))) + continue; + if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) { + int mx = 0, my = 0; + map_mouse(event.motion.x, event.motion.y, mx, my); + pending_mouse_move = std::pair{mx, my}; + continue; + } + // Preserve event order at button/key/focus boundaries while collapsing + // high-frequency motion into the most recent position. + flush_mouse_move(); +#endif + if (event.type == SDL_EVENT_QUIT) return false; + if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) { + int ew = 0, eh = 0; + SDL_GetWindowSize(window_, &ew, &eh); + touch_controller.update_screen_size(int(logical_width()), int(logical_height()), ui_safe_inset()); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && ew > 0 && eh > 0) { + PythonBoot::SetUISafeInset(ui_safe_inset()); + PythonBoot::SetUISize(int(logical_width()), int(logical_height())); + } +#endif + recreate_swapchain(); + continue; + } + if (event.type == SDL_EVENT_FINGER_DOWN) { + touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } else if (event.type == SDL_EVENT_FINGER_MOTION) { + touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) { + touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client) { + if (!hardware_cursor_enabled_ && + (event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) { + std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n"); + SDL_HideCursor(); + os_cursor_hidden_ = true; + } else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) { + SDL_CaptureMouse(false); + PythonBoot::UIMouseButton(2, false, last_mx_, last_my_); + PythonBoot::UIMouseButton(3, false, last_mx_, last_my_); + PythonBoot::UIMouseButton(1, false, last_mx_, last_my_); + } + if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) { + const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN; + int btn = 0; + if (event.button.button == SDL_BUTTON_LEFT) btn = 1; + else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2; + else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3; + if (btn) { + int mx = 0, my = 0; + map_mouse(event.button.x, event.button.y, mx, my); + last_mx_ = mx; + last_my_ = my; + if (touch_controller.is_enabled()) { + if (touch_controller.on_mouse_button(btn, pressed, mx, my)) { + continue; + } + } + SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0); + PythonBoot::UIMouseButton(btn, pressed, mx, my); + } + } else if (event.type == SDL_EVENT_MOUSE_WHEEL) { + PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f)); + } else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) { + const bool pressed = event.type == SDL_EVENT_KEY_DOWN; + // Android back = ESC: closes the top window, or opens the system menu (game.py). + if (event.key.scancode == SDL_SCANCODE_AC_BACK) event.key.scancode = SDL_SCANCODE_ESCAPE; + if (pressed) { + if (const int vk = sdl_scancode_to_vk(event.key.scancode)) + PythonBoot::UIIMEKeyDown(vk); + if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8); + else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9); + else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER) + PythonBoot::UIChar(13); + else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27); + } + if (!event.key.repeat) { + if (const int dik = sdl_scancode_to_dik(event.key.scancode)) + PythonBoot::UIKey(dik, pressed); + } + } else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) { + const auto* s = reinterpret_cast(event.text.text); + while (*s) { + unsigned cp = 0; + if (*s < 0x80u) { + cp = *s++; + } else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) { + cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu); + s += 2; + } else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) { + cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu); + s += 3; + } else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) { + cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu); + s += 4; + } else { + ++s; + continue; + } + if (cp >= 32u && cp != 127u) + PythonBoot::UIChar(cp); + } + } + } +#else + (void)forward_to_live_client; +#endif + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + flush_mouse_move(); +#endif + touch_controller.update(); + if (touch_controller.is_enabled()) + sync_screen_keyboard(); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) { + SDL_HideCursor(); + os_cursor_hidden_ = true; + } +#endif + return true; +} + +void VulkanWindow::reset_timings() { + finish_gpu_timings(); + timings_ = {}; + timed_frames_ = 0; + upload_count_ = 0; + uploaded_bytes_ = 0; + texture_upload_count_ = 0; + texture_uploaded_bytes_ = 0; +} + +void VulkanWindow::finish_gpu_timings() { + for (auto& slot : frames_) { + if (!slot.has_pending_query) continue; + check(vkWaitForFences(device_, 1, &slot.fence, VK_TRUE, UINT64_MAX), "vkWaitForFences finish"); + collect_pending_gpu_timestamp(slot); + } +} + +void VulkanWindow::set_frames_in_flight(std::uint32_t count) { + frames_in_flight_ = std::clamp(count, 1, kMaxFramesInFlight); +} + +void VulkanWindow::render( + const std::vector& draws, + const std::unordered_map>& capture_textures, + std::uint32_t ui_width, + std::uint32_t ui_height, + const std::vector& ui_commands) { + if (extent_.width == 0 || extent_.height == 0) return; + const auto start = std::chrono::steady_clock::now(); + f_ = &frames_[frame_slot_ % frames_in_flight_]; + check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences"); + collect_pending_gpu_timestamp(*f_); + reset_staging(*f_); + const auto fenced = std::chrono::steady_clock::now(); + prune_textures(); + std::uint32_t image_index = 0; + const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, f_->acquire, VK_NULL_HANDLE, &image_index); + if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { + recreate_swapchain(); + return; + } + if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR"); + const auto synchronized = std::chrono::steady_clock::now(); + ++frame_number_; + ++timed_frames_; + while (rendered_.size() < swapchain_images_.size()) { + VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + VkSemaphore semaphore = VK_NULL_HANDLE; + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &semaphore), "vkCreateSemaphore rendered"); + rendered_.push_back(semaphore); + } + // Texture uploads of this frame are recorded ahead of the render pass (upload_texture). + check(vkResetCommandBuffer(f_->command, 0), "vkResetCommandBuffer"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(f_->command, &begin), "vkBeginCommandBuffer"); + vkCmdResetQueryPool(f_->command, query_pool_, f_->query_base, 2); + vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, f_->query_base); + f_->recording = true; + + std::vector prepared_draws; + prepared_draws.reserve(draws.size()); + // Draws into offscreen render targets (the shadow map), in recorded order, drawn before the + // back-buffer pass that samples them. + std::vector offscreen_draws; + std::unordered_set active_keys; + std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; + std::uint32_t bone_cursor = 0; + std::size_t required_bones = 0; + for (const auto& draw : draws) { + if (!draw.bone_matrices.empty()) { + if (draw.bone_matrices.size() % 16) + throw std::runtime_error("invalid bone palette length"); + required_bones += draw.bone_matrices.size() / 16; + } + } + ensure_bone_capacity(required_bones); + + for (const auto& draw : draws) { + RenderTarget* target = nullptr; + if (draw.render_target) { + target = get_render_target( + Render3DRenderTargetName(draw.render_target, draw.target_width, draw.target_height)); + if (!target) continue; + } + if (draw.clear_flags && target) { + PreparedDraw clear{}; + clear.target = target; + clear.clear_flags = draw.clear_flags; + const auto color = unpack_argb(draw.clear_color); + clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; + clear.clear_depth.depthStencil = {draw.clear_z, 0}; + clear.clear_rect = {{0, 0}, {target->width, target->height}}; + offscreen_draws.push_back(clear); + continue; + } + if (draw.clear_flags) { + PreparedDraw clear{}; + clear.clear_flags = draw.clear_flags; + const auto color = unpack_argb(draw.clear_color); + clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; + clear.clear_depth.depthStencil = {draw.clear_z, 0}; + // D3D8 Clear with no rectangles clears the current viewport. + const auto viewport = draw_viewport(draw); + const float x0 = std::clamp(viewport.x, 0.0f, float(extent_.width)); + const float y0 = std::clamp(viewport.y, 0.0f, float(extent_.height)); + const float x1 = std::clamp(viewport.x + viewport.width, 0.0f, float(extent_.width)); + const float y1 = std::clamp(viewport.y + viewport.height, 0.0f, float(extent_.height)); + clear.clear_rect = {{std::int32_t(x0), std::int32_t(y0)}, + {std::uint32_t(x1 - x0), std::uint32_t(y1 - y0)}}; + prepared_draws.push_back(clear); + continue; + } + if (draw.positions.empty() || draw.indices.empty()) continue; + const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw); + const GeometryId id{draw.geometry_key, signature}; + auto [it, inserted] = geometries_.try_emplace(id); + auto& geometry = it->second; + if (inserted) { + upload_geometry(draw, geometry); + } else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) { + throw std::runtime_error("geometry key/revision reused with a different size"); + } + geometry.last_used_frame = frame_number_; + if (draw.geometry_key) active_keys.insert(draw.geometry_key); + + float skin_offset_encoded = 0.0f; + if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && f_->bone_mapped) { + const auto bone_count = static_cast(draw.bone_matrices.size() / 16); + if (std::size_t(bone_cursor) + bone_count <= f_->bone_capacity) { + std::memcpy( + f_->bone_mapped + std::size_t(bone_cursor) * 16, + draw.bone_matrices.data(), + draw.bone_matrices.size() * sizeof(float)); + skin_offset_encoded = float(bone_cursor + 1u); + bone_cursor += bone_count; + ++skinned_draw_count; + } else throw std::runtime_error("bone palette capacity exceeded"); + } + + const bool is_alpha = draw.alpha_blend != 0; + std::uint8_t cull = 0; + // D3D8 cull mode names describe the winding to discard. With the + // shader's Y flip, clockwise is Vulkan's front-facing winding. + if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise + else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise + const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0; + std::uint8_t blend = 0; + if (is_alpha) { + if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 && + draw.dest_blend >= 1 && draw.dest_blend <= 11) { + blend = static_cast((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4)); + } else { + blend = 1; + } + } + const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, + static_cast(draw.z_func), target != nullptr); + // D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples. + const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1; + const auto descriptor = get_texture_descriptor( + draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw); + + auto constants = make_push_constants(draw, skin_offset_encoded); + if (draw.pretransformed) { + const float screen_width = float(logical_width()); + const float screen_height = float(logical_height()); + // Screen pixels (y down) to D3D NDC (y up); the shader's Y flip then maps to Vulkan. + constants.mvp = {2.0f / screen_width, 0, 0, 0, + 0, -2.0f / screen_height, 0, 0, + 0, 0, 1, 0, + -1, 1, 0, 1}; + } + PreparedDraw prepared_draw{}; + prepared_draw.target = target; + prepared_draw.geometry = &geometry; + prepared_draw.pipeline = pipeline; + prepared_draw.descriptor = descriptor; + prepared_draw.constants = constants; + prepared_draw.fixed_state = make_fixed_function_state(draw); + // XYZRHW vertices are already in screen space and bypass the viewport transform. + prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw); + if (draw.pretransformed && draw.ui_offset_x != 0) + prepared_draw.viewport.x += draw.ui_offset_x * float(extent_.width) / float(logical_width()); + if (target) { + // D3DVIEWPORT8 in the target's own pixels. + prepared_draw.viewport = draw.viewport[2] > 0 && draw.viewport[3] > 0 + ? VkViewport{draw.viewport[0], draw.viewport[1], draw.viewport[2], draw.viewport[3], + draw.viewport_z[0], draw.viewport_z[1]} + : VkViewport{0, 0, float(target->width), float(target->height), 0, 1}; + offscreen_draws.push_back(prepared_draw); + } else prepared_draws.push_back(prepared_draw); + vertex_count += geometry.vertex_count; + index_count += geometry.index_count; + } + for (auto it = geometries_.begin(); it != geometries_.end();) { + const auto age = frame_number_ - it->second.last_used_frame; + const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key); + if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) { + release_geometry(it->second); + it = geometries_.erase(it); + } else ++it; + } + + std::vector ui_batches; + std::size_t ui_quad_count = 0; + if (f_->ui_mapped) { + const uint32_t target_ui_w = ui_width ? ui_width : 960; + const uint32_t target_ui_h = ui_height ? ui_height : 640; + update_fps_counter(); + if (touch_controller.is_enabled() || show_fps_) { + std::vector combined_ui = ui_commands; + if (touch_controller.is_enabled()) { + touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h), ui_safe_inset()); + touch_controller.append_ui_commands(combined_ui); + } + if (show_fps_ && fps_ >= 0) { + // Top left, inside the safe area and clear of 40250's corner icon. + const float inset = float(std::min(ui_safe_inset(), int(target_ui_w) / 8)); + TouchController::draw_segment_text(combined_ui, "FPS " + std::to_string(fps_), + inset + 40.0f, 8.0f, 12.0f, 0xFF40FF40); + } + if (!combined_ui.empty()) { + build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count); + } + } else if (!ui_commands.empty()) { + build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count); + } + } + + ensure_state_capacity(prepared_draws.size() + offscreen_draws.size() + ui_batches.size()); + std::size_t state_index = 0; + for (auto& draw : offscreen_draws) { + if (!draw.geometry) continue; + draw.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + ++state_index; + } + last_offscreen_draw_count_ = offscreen_draws.size(); + for (auto& draw : prepared_draws) { + if (!draw.geometry) continue; + draw.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + ++state_index; + } + const FixedFunctionState ui_state{}; + for (auto& batch : ui_batches) { + batch.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); + ++state_index; + } + + const auto prepared = std::chrono::steady_clock::now(); + f_->recording = false; + check(vkResetFences(device_, 1, &f_->fence), "vkResetFences"); + + VkClearValue clears[3]{}; + // 40250 clears the back buffer to 0xff000000 once at device creation and afterwards + // only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer). + clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}}; + clears[1].depthStencil = {1.0f, 0}; + VkPipeline current_pipeline = VK_NULL_HANDLE; + VkDescriptorSet current_descriptor = VK_NULL_HANDLE; + + VkViewport current_viewport{-1, -1, -1, -1, -1, -1}; + auto set_viewport = [&](const VkViewport& viewport) { + if (std::memcmp(&viewport, ¤t_viewport, sizeof(viewport)) == 0) return; + vkCmdSetViewport(f_->command, 0, 1, &viewport); + current_viewport = viewport; + }; + + // One prepared draw or Clear inside the current render pass. + auto record_prepared = [&](const PreparedDraw& prepared_draw) { + if (!prepared_draw.geometry) { + VkClearAttachment attachments[2]{}; + std::uint32_t attachment_count = 0; + if (prepared_draw.clear_flags & 1u) { // D3DCLEAR_TARGET + attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + attachments[attachment_count].colorAttachment = 0; + attachments[attachment_count].clearValue = prepared_draw.clear_color; + ++attachment_count; + } + if (prepared_draw.clear_flags & 2u) { // D3DCLEAR_ZBUFFER + attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; + attachments[attachment_count].clearValue = prepared_draw.clear_depth; + ++attachment_count; + } + const VkClearRect rect{prepared_draw.clear_rect, 0, 1}; + if (attachment_count && rect.rect.extent.width && rect.rect.extent.height) + vkCmdClearAttachments(f_->command, attachment_count, attachments, 1, &rect); + return; + } + set_viewport(prepared_draw.viewport); + if (prepared_draw.pipeline != current_pipeline) { + vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); + current_pipeline = prepared_draw.pipeline; + } + if (prepared_draw.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); + current_descriptor = prepared_draw.descriptor; + } + vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &f_->bone_descriptor_set, 1, &prepared_draw.state_offset); + const auto& geometry = *prepared_draw.geometry; + const VkDeviceSize vertex_offset = 0; + vkCmdBindVertexBuffers(f_->command, 0, 1, &geometry.buffer, &vertex_offset); + vkCmdBindIndexBuffer(f_->command, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); + vkCmdPushConstants( + f_->command, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(prepared_draw.constants), + &prepared_draw.constants); + vkCmdDrawIndexed(f_->command, geometry.index_count, 1, 0, 0, 0); + }; + + // Render targets first used this frame (drawn or only sampled) start cleared. + for (auto& entry : render_targets_) + if (!entry.second.initialized) initialize_render_target(entry.second); + // Offscreen passes: one per run of draws into the same target. + for (std::size_t i = 0; i < offscreen_draws.size();) { + RenderTarget* target = offscreen_draws[i].target; + VkRenderPassBeginInfo offscreen_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + offscreen_begin.renderPass = offscreen_pass_; + offscreen_begin.framebuffer = target->framebuffer; + offscreen_begin.renderArea.extent = {target->width, target->height}; + vkCmdBeginRenderPass(f_->command, &offscreen_begin, VK_SUBPASS_CONTENTS_INLINE); + for (; i < offscreen_draws.size() && offscreen_draws[i].target == target; ++i) + record_prepared(offscreen_draws[i]); + vkCmdEndRenderPass(f_->command); + } + + VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); + pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears; + vkCmdBeginRenderPass(f_->command, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); + + auto record_ui_pass = [&](bool behind_3d) { + bool ui_bound = false; + set_viewport(full_viewport()); + for (const auto& batch : ui_batches) { + if (batch.behind_3d != behind_3d) continue; + if (!ui_bound) { + const VkDeviceSize v_offset = 0; + vkCmdBindVertexBuffers(f_->command, 0, 1, &f_->ui_buffer, &v_offset); + vkCmdBindIndexBuffer(f_->command, f_->ui_buffer, f_->ui_vertex_bytes, VK_INDEX_TYPE_UINT32); + ui_bound = true; + } + if (batch.pipeline != current_pipeline) { + vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); + current_pipeline = batch.pipeline; + } + if (batch.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); + current_descriptor = batch.descriptor; + } + vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &f_->bone_descriptor_set, 1, &batch.state_offset); + vkCmdPushConstants( + f_->command, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(batch.constants), + &batch.constants); + vkCmdDrawIndexed(f_->command, batch.index_count, 1, batch.first_index, 0, 0); + } + }; + + record_ui_pass(true); + + for (const auto& prepared_draw : prepared_draws) record_prepared(prepared_draw); + + record_ui_pass(false); + + vkCmdEndRenderPass(f_->command); + VkBuffer readback = VK_NULL_HANDLE; + VkDeviceMemory readback_memory = VK_NULL_HANDLE; + const bool screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_; + if (screenshot) { + const VkDeviceSize bytes = VkDeviceSize(extent_.width) * extent_.height * 4; + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = bytes; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; + check(vkCreateBuffer(device_, &info, nullptr, &readback), "vkCreateBuffer readback"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, readback, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &readback_memory), "vkAllocateMemory readback"); + check(vkBindBufferMemory(device_, readback, readback_memory, 0), "vkBindBufferMemory readback"); + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.image = swapchain_images_.at(image_index); + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + VkBufferImageCopy region{}; + region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; + region.imageExtent = {extent_.width, extent_.height, 1}; + vkCmdCopyImageToBuffer(f_->command, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, ®ion); + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = 0; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + } + // Debug: MT_DUMP_RT=prefix writes each render target of the screenshot frame to _.pgm. + struct RtDump { VkBuffer buffer; VkDeviceMemory memory; std::uint32_t w, h; }; + std::vector rt_dumps; + const char* dump_rt = std::getenv("MT_DUMP_RT"); + if (screenshot && dump_rt && *dump_rt) { + const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; + for (auto& entry : render_targets_) { + RenderTarget& rt = entry.second; + RtDump dump{VK_NULL_HANDLE, VK_NULL_HANDLE, rt.width, rt.height}; + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = VkDeviceSize(rt.width) * rt.height * texel; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; + check(vkCreateBuffer(device_, &info, nullptr, &dump.buffer), "vkCreateBuffer rt dump"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, dump.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &dump.memory), "vkAllocateMemory rt dump"); + check(vkBindBufferMemory(device_, dump.buffer, dump.memory, 0), "vkBindBufferMemory rt dump"); + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_SHADER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.image = rt.texture.image; + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + VkBufferImageCopy region{}; + region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; + region.imageExtent = {rt.width, rt.height, 1}; + vkCmdCopyImageToBuffer(f_->command, rt.texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dump.buffer, 1, ®ion); + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + rt_dumps.push_back(dump); + } + } + vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, f_->query_base + 1); + check(vkEndCommandBuffer(f_->command), "vkEndCommandBuffer"); + f_->has_pending_query = true; + + const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &f_->acquire; submit.pWaitDstStageMask = &stage; + submit.commandBufferCount = 1; submit.pCommandBuffers = &f_->command; + submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_[image_index]; + check(vkQueueSubmit(queue_, 1, &submit, f_->fence), "vkQueueSubmit"); + ++frame_slot_; + if (screenshot) { + check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences readback"); + void* mapped = nullptr; + check(vkMapMemory(device_, readback_memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback"); + write_bmp(screenshot_path_, static_cast(mapped)); + vkUnmapMemory(device_, readback_memory); + const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; + for (std::size_t n = 0; n < rt_dumps.size(); ++n) { + const auto& dump = rt_dumps[n]; + void* data = nullptr; + check(vkMapMemory(device_, dump.memory, 0, VK_WHOLE_SIZE, 0, &data), "vkMapMemory rt dump"); + const auto* bytes = static_cast(data); + std::ofstream out(std::string(dump_rt) + "_" + std::to_string(n) + ".pgm", std::ios::binary); + out << "P5\n" << dump.w << " " << dump.h << "\n255\n"; + for (std::size_t i = 0; i < std::size_t(dump.w) * dump.h; ++i) { + std::uint8_t g = texel == 2 ? std::uint8_t(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31) : bytes[i * 4]; + out.put(char(g)); + } + vkUnmapMemory(device_, dump.memory); + vkDestroyBuffer(device_, dump.buffer, nullptr); + vkFreeMemory(device_, dump.memory, nullptr); + } + vkDestroyBuffer(device_, readback, nullptr); + vkFreeMemory(device_, readback_memory, nullptr); + } + const auto submitted = std::chrono::steady_clock::now(); + VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; + present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_[image_index]; + present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; + const auto presented = vkQueuePresentKHR(queue_, &present); + // An Android surface can continuously report SUBOPTIMAL while the app deliberately + // uses the identity transform. The image is still presentable; rebuilding every frame + // stalls rendering without changing that status. + if (presented == VK_ERROR_OUT_OF_DATE_KHR) { + recreate_swapchain(); + } else if (presented != VK_SUCCESS && presented != VK_SUBOPTIMAL_KHR) { + check(presented, "vkQueuePresentKHR"); + } + const auto presented_at = std::chrono::steady_clock::now(); + const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); + last_sync_ms_ = std::chrono::duration(synchronized - start).count(); + timings_.sync_ms += last_sync_ms_; + timings_.fence_ms += std::chrono::duration(fenced - start).count(); + timings_.prepare_ms += prepare_ms; + if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; + timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); + timings_.present_ms += std::chrono::duration(presented_at - submitted).count(); + last_draw_count_ = prepared_draws.size(); + last_skinned_draw_count_ = skinned_draw_count; + last_ui_batch_count_ = ui_batches.size(); + last_ui_quad_count_ = ui_quad_count; + last_vertex_count_ = vertex_count; + last_index_count_ = index_count; +} + +const char* VulkanWindow::present_mode_name() const { + switch (present_mode_) { + case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE"; + case VK_PRESENT_MODE_MAILBOX_KHR: return "MAILBOX"; + default: return "FIFO"; + } +} + +native_perf::RendererTotals VulkanWindow::perf_totals() const { + native_perf::RendererTotals t; + t.sync_ms = timings_.sync_ms; + t.fence_ms = timings_.fence_ms; + t.prepare_ms = timings_.prepare_ms; + t.geometry_upload_ms = timings_.geometry_upload_ms; + t.texture_upload_ms = timings_.texture_upload_ms; + t.submit_ms = timings_.submit_ms; + t.present_ms = timings_.present_ms; + t.gpu_ms = timings_.gpu_ms; + t.gpu_samples = timings_.gpu_samples; + t.uploads = upload_count_; + t.uploaded_bytes = uploaded_bytes_; + t.texture_uploads = texture_upload_count_; + t.texture_bytes = texture_uploaded_bytes_; + t.draws = last_draw_count_; + t.skinned_draws = last_skinned_draw_count_; + t.ui_batches = last_ui_batch_count_; + t.vertices = last_vertex_count_; + t.last_texture = &last_texture_upload_name_; + return t; +} + +void VulkanWindow::recreate_swapchain() { + int w = 0, h = 0; + SDL_GetWindowSizeInPixels(window_, &w, &h); + if (w <= 0 || h <= 0) return; + vkDeviceWaitIdle(device_); + for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); + pipelines_.clear(); + for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr); + framebuffers_.clear(); + if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; } + if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; } + if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; } + destroy_msaa_color(); + for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); + image_views_.clear(); + if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; } + create_swapchain(); + create_framebuffers(); + get_pipeline(0, 2, 0); +} + +void VulkanWindow::collect_pending_gpu_timestamp(FrameSlot& slot) { + if (!slot.has_pending_query) return; + slot.has_pending_query = false; + std::uint64_t timestamps[2] = {}; + const VkResult res = vkGetQueryPoolResults( + device_, query_pool_, slot.query_base, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), + VK_QUERY_RESULT_64_BIT); + if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) { + const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_); + timings_.gpu_ms += ns * 1e-6; + ++timings_.gpu_samples; + } +} + +void VulkanWindow::build_ui_batches( + std::uint32_t ui_width, + std::uint32_t ui_height, + const std::vector& commands, + const std::unordered_map>& capture_textures, + std::vector& batches, + std::size_t& out_quad_count) { + ensure_ui_capacity(commands.size()); + auto* vertices = reinterpret_cast(f_->ui_mapped); + auto* indices = reinterpret_cast(f_->ui_mapped + f_->ui_vertex_bytes); + std::uint32_t v_count = 0; + std::uint32_t i_count = 0; + const float inv_w = 2.0f / float(ui_width); + const float inv_h = 2.0f / float(ui_height); + + for (const auto& cmd : commands) { + if (v_count + 4 > f_->ui_vertex_capacity || i_count + 6 > f_->ui_index_capacity) + throw std::runtime_error("UI command buffer capacity exceeded"); + if (cmd.kind == UIRenderCommand::Text) continue; + + float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2; + float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev; + const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1; + + float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; + std::array top_color = unpack_argb(cmd.argb); + std::array bot_color = + cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color; + if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue; + + if (cmd.kind == UIRenderCommand::Line) { + const float dx = x2 - x1, dy = y2 - y1; + const float len = std::sqrt(dx * dx + dy * dy); + if (len < 0.1f) continue; + const float nx = -dy / len * 0.5f; + const float ny = dx / len * 0.5f; + qx[0] = x1 + nx; qy[0] = y1 + ny; + qx[1] = x2 + nx; qy[1] = y2 + ny; + qx[2] = x1 - nx; qy[2] = y1 - ny; + qx[3] = x2 - nx; qy[3] = y2 - ny; + for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; } + } else if (cmd.kind == UIRenderCommand::Image && cmd.quad) { + const bool axis_aligned = + std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f && + std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f && + std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f && + std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f; + for (int k = 0; k < 4; ++k) { + mu[k] = cmd.mu[k]; + mv[k] = cmd.mv[k]; + } + if (has_clip && axis_aligned) { + const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3]; + const float cx1 = std::max(ox1, cmd.clip_x1); + const float cy1 = std::max(oy1, cmd.clip_y1); + const float cx2 = std::min(ox2, cmd.clip_x2); + const float cy2 = std::min(oy2, cmd.clip_y2); + if (cx2 <= cx1 || cy2 <= cy1) continue; + float msu = cmd.mu[0], meu = cmd.mu[1]; + float msv = cmd.mv[0], mev = cmd.mv[2]; + if (ox2 > ox1) { + const float t0 = (cx1 - ox1) / (ox2 - ox1); + const float t1 = (cx2 - ox1) / (ox2 - ox1); + const float du = eu - su; + su = cmd.su + du * t0; + eu = cmd.su + du * t1; + const float mdu = meu - msu; + msu = cmd.mu[0] + mdu * t0; + meu = cmd.mu[0] + mdu * t1; + } + if (oy2 > oy1) { + const float t0 = (cy1 - oy1) / (oy2 - oy1); + const float t1 = (cy2 - oy1) / (oy2 - oy1); + const float dv = ev - sv; + sv = cmd.sv + dv * t0; + ev = cmd.sv + dv * t1; + const float mdv = mev - msv; + msv = cmd.mv[0] + mdv * t0; + mev = cmd.mv[0] + mdv * t1; + } + qx[0] = cx1; qy[0] = cy1; + qx[1] = cx2; qy[1] = cy1; + qx[2] = cx1; qy[2] = cy2; + qx[3] = cx2; qy[3] = cy2; + mu[0] = msu; mv[0] = msv; + mu[1] = meu; mv[1] = msv; + mu[2] = msu; mv[2] = mev; + mu[3] = meu; mv[3] = mev; + } else { + if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) + continue; + for (int k = 0; k < 4; ++k) { + qx[k] = cmd.qx[k]; + qy[k] = cmd.qy[k]; + } + } + u[0] = su; v[0] = sv; + u[1] = eu; v[1] = sv; + u[2] = su; v[2] = ev; + u[3] = eu; v[3] = ev; + } else if (cmd.kind == UIRenderCommand::Bar && cmd.quad) { + // An untextured polygon piece (CPythonGraphic::RenderCoolTimeBox's fan triangles). + if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) + continue; + for (int k = 0; k < 4; ++k) { + qx[k] = cmd.qx[k]; + qy[k] = cmd.qy[k]; + u[k] = 0.0f; + v[k] = 0.0f; + } + } else { + if (has_clip) { + x1 = std::max(x1, cmd.clip_x1); + y1 = std::max(y1, cmd.clip_y1); + x2 = std::min(x2, cmd.clip_x2); + y2 = std::min(y2, cmd.clip_y2); + } + if (x2 <= x1 || y2 <= y1) continue; + qx[0] = x1; qy[0] = y1; + qx[1] = x2; qy[1] = y1; + qx[2] = x1; qy[2] = y2; + qx[3] = x2; qy[3] = y2; + u[0] = su; v[0] = sv; + u[1] = eu; v[1] = sv; + u[2] = su; v[2] = ev; + u[3] = eu; v[3] = ev; + } + + const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string(); + const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string(); + const bool is_masked = !mask_name.empty(); + // CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use + // INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR. + const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image + ? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1)) + : 1; + const VkPipeline pipeline = get_pipeline(0, 0, blend_mode); + const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures); + + for (int k = 0; k < 4; ++k) { + Vertex& vert = vertices[v_count + k]; + vert.position[0] = qx[k] * inv_w - 1.0f; + vert.position[1] = 1.0f - qy[k] * inv_h; + vert.position[2] = 0.0f; + vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f; + vert.uv[0] = u[k]; vert.uv[1] = v[k]; + const auto& col = (k < 2) ? top_color : bot_color; + vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3]; + vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0; + vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f; + vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k]; + vert.rhw = 1.0f; + } + indices[i_count + 0] = v_count + 0; + indices[i_count + 1] = v_count + 1; + indices[i_count + 2] = v_count + 2; + indices[i_count + 3] = v_count + 2; + indices[i_count + 4] = v_count + 1; + indices[i_count + 3 + 2] = v_count + 3; + + const float mask_flag = is_masked ? 1.0f : 0.0f; + if (!batches.empty() && + batches.back().behind_3d == cmd.behind_3d && + batches.back().pipeline == pipeline && + batches.back().descriptor == descriptor && + batches.back().constants.params[2] == mask_flag) { + batches.back().index_count += 6; + } else { + UiBatch batch{}; + batch.first_index = i_count; + batch.index_count = 6; + batch.pipeline = pipeline; + batch.descriptor = descriptor; + batch.behind_3d = cmd.behind_3d; + batch.constants.mvp = kIdentityMatrix; + batch.constants.params = {0.0f, 0.0f, mask_flag, 0.0f}; + batches.push_back(batch); + } + v_count += 4; + i_count += 6; + ++out_quad_count; + } +} + +std::uint32_t VulkanWindow::find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const { + VkPhysicalDeviceMemoryProperties properties{}; + vkGetPhysicalDeviceMemoryProperties(physical_, &properties); + for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i) + if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags) + return i; + throw std::runtime_error("no matching Vulkan memory type"); +} + +void VulkanWindow::select_device() { + std::uint32_t count = 0; + check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count"); + if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK"); + std::vector candidates(count); + check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices"); + for (auto candidate : candidates) { + std::uint32_t family_count = 0; + vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr); + std::vector families(family_count); + vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data()); + for (std::uint32_t i = 0; i < family_count; ++i) { + VkBool32 present = VK_FALSE; + check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support"); + if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) { + physical_ = candidate; queue_family_ = i; break; + } + } + if (physical_) break; + } + if (!physical_) throw std::runtime_error("no graphics/present queue family"); + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + device_name_ = properties.deviceName; + timestamp_period_ns_ = properties.limits.timestampPeriod; + VkPhysicalDeviceFeatures supported_features{}; + vkGetPhysicalDeviceFeatures(physical_, &supported_features); + VkPhysicalDeviceFeatures enabled_features{}; + if (supported_features.samplerAnisotropy) { + enabled_features.samplerAnisotropy = VK_TRUE; + max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy); + } + // Multisampled 3D edges; MT_MSAA=1 keeps the single-sample path, 2/4/8 request a count. + { + int wanted = 4; + if (const char* env = std::getenv("MT_MSAA")) wanted = std::max(1, std::atoi(env)); + const VkSampleCountFlags supported = + properties.limits.framebufferColorSampleCounts & properties.limits.framebufferDepthSampleCounts; + samples_ = VK_SAMPLE_COUNT_1_BIT; + for (VkSampleCountFlagBits c : {VK_SAMPLE_COUNT_8_BIT, VK_SAMPLE_COUNT_4_BIT, VK_SAMPLE_COUNT_2_BIT}) + if (int(c) <= wanted && (supported & c)) { samples_ = c; break; } + } + std::uint32_t extension_count = 0; + check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count"); + std::vector available(extension_count); + check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions"); + std::vector extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME}; + constexpr const char* portability_subset = "VK_KHR_portability_subset"; + for (const auto& extension : available) + if (std::strcmp(extension.extensionName, portability_subset) == 0) + extensions.push_back(portability_subset); + const float priority = 1.0f; + VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; + queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority; + VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; + device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info; + device_info.enabledExtensionCount = static_cast(extensions.size()); + device_info.ppEnabledExtensionNames = extensions.data(); + device_info.pEnabledFeatures = &enabled_features; + check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice"); + vkGetDeviceQueue(device_, queue_family_, 0, &queue_); +} + +void VulkanWindow::create_swapchain() { + VkSurfaceCapabilitiesKHR capabilities{}; + check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities"); + std::uint32_t count = 0; + check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count"); + if (!count) throw std::runtime_error("surface has no formats"); + std::vector formats(count); + check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats"); + VkSurfaceFormatKHR format = formats.front(); + for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate; + swapchain_format_ = format.format; + int width = 0, height = 0; + SDL_GetWindowSizeInPixels(window_, &width, &height); + if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent; + else { + extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width); + extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height); + } + + present_mode_ = VK_PRESENT_MODE_FIFO_KHR; + if (!vsync_) { + std::uint32_t mode_count = 0; + check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count"); + std::vector modes(mode_count); + if (mode_count) + check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes"); + if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end()) + present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR; + else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end()) + present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR; + } + + VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR}; + info.surface = surface_; + info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX); + info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_; + info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | + (capabilities.supportedUsageFlags & VK_IMAGE_USAGE_TRANSFER_SRC_BIT); + info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; + // Android surfaces can report a quarter-turn transform even when SDL's window is + // already landscape. Rendering in SDL window coordinates and requesting that transform + // again rotates the entire game image on presentation. + info.preTransform = (capabilities.supportedTransforms & VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR) + ? VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR : capabilities.currentTransform; +#ifdef __ANDROID__ + SDL_Log("swapchain: window=%dx%d extent=%ux%u currentTransform=%u preTransform=%u", + width, height, extent_.width, extent_.height, + unsigned(capabilities.currentTransform), unsigned(info.preTransform)); +#endif + info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_; + info.clipped = VK_TRUE; + check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR"); + check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count"); + std::vector images(count); + check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images"); + swapchain_images_ = images; + for (auto image : images) { + VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; + view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; + VkImageView handle = VK_NULL_HANDLE; + check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView"); + image_views_.push_back(handle); + } + VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + depth_info.imageType = VK_IMAGE_TYPE_2D; + depth_info.format = VK_FORMAT_D32_SFLOAT; + depth_info.extent = {extent_.width, extent_.height, 1}; + depth_info.mipLevels = 1; depth_info.arrayLayers = 1; + depth_info.samples = samples_; + depth_info.tiling = VK_IMAGE_TILING_OPTIMAL; + depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT; + depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth"); + VkMemoryRequirements depth_reqs{}; + vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs); + VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + depth_alloc.allocationSize = depth_reqs.size; + depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth"); + check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth"); + VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D; + depth_view.format = VK_FORMAT_D32_SFLOAT; + depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; + depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1; + check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth"); + if (samples_ != VK_SAMPLE_COUNT_1_BIT) create_msaa_color(); +} + +void VulkanWindow::create_msaa_color() { + VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + info.imageType = VK_IMAGE_TYPE_2D; + info.format = swapchain_format_; + info.extent = {extent_.width, extent_.height, 1}; + info.mipLevels = 1; info.arrayLayers = 1; + info.samples = samples_; + info.tiling = VK_IMAGE_TILING_OPTIMAL; + info.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSIENT_ATTACHMENT_BIT; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &info, nullptr, &msaa_image_), "vkCreateImage msaa color"); + VkMemoryRequirements reqs{}; + vkGetImageMemoryRequirements(device_, msaa_image_, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + try { + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_LAZILY_ALLOCATED_BIT); + } catch (const std::runtime_error&) { + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + } + check(vkAllocateMemory(device_, &alloc, nullptr, &msaa_memory_), "vkAllocateMemory msaa color"); + check(vkBindImageMemory(device_, msaa_image_, msaa_memory_, 0), "vkBindImageMemory msaa color"); + VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view.image = msaa_image_; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; + view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; + check(vkCreateImageView(device_, &view, nullptr, &msaa_view_), "vkCreateImageView msaa color"); +} + +void VulkanWindow::destroy_msaa_color() { + if (msaa_view_) { vkDestroyImageView(device_, msaa_view_, nullptr); msaa_view_ = VK_NULL_HANDLE; } + if (msaa_image_) { vkDestroyImage(device_, msaa_image_, nullptr); msaa_image_ = VK_NULL_HANDLE; } + if (msaa_memory_) { vkFreeMemory(device_, msaa_memory_, nullptr); msaa_memory_ = VK_NULL_HANDLE; } +} + +void VulkanWindow::create_framebuffers() { + for (auto view : image_views_) { + // Single-sample: {swapchain, depth}. MSAA: {msaa colour, depth, swapchain resolve}. + VkImageView fb_attachments[3] = {view, depth_view_, VK_NULL_HANDLE}; + std::uint32_t count = 2; + if (samples_ != VK_SAMPLE_COUNT_1_BIT) { + fb_attachments[0] = msaa_view_; + fb_attachments[2] = view; + count = 3; + } + VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + framebuffer.renderPass = pass_; framebuffer.attachmentCount = count; framebuffer.pAttachments = fb_attachments; + framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; + VkFramebuffer handle = VK_NULL_HANDLE; + check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer"); + framebuffers_.push_back(handle); + } +} + +void VulkanWindow::create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) { + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = size; + info.usage = usage; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host"); + VkMemoryRequirements reqs{}; + vkGetBufferMemoryRequirements(device_, buffer, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + alloc.memoryTypeIndex = find_memory_type( + reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host"); + check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host"); + check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host"); +} + +void VulkanWindow::create_descriptors_and_buffers() { + VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + sampler_info.magFilter = VK_FILTER_LINEAR; + sampler_info.minFilter = VK_FILTER_LINEAR; + sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; + sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.minLod = 0.0f; + sampler_info.maxLod = VK_LOD_CLAMP_NONE; + if (max_anisotropy_ > 1.0f) { + sampler_info.anisotropyEnable = VK_TRUE; + sampler_info.maxAnisotropy = max_anisotropy_; + } + check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler"); + + VkSamplerCreateInfo clamp_info = sampler_info; + clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp"); + + VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + ui_sampler_info.magFilter = VK_FILTER_LINEAR; + ui_sampler_info.minFilter = VK_FILTER_LINEAR; + ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST; + ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.minLod = 0.0f; + ui_sampler_info.maxLod = 0.0f; + ui_sampler_info.anisotropyEnable = VK_FALSE; + ui_sampler_info.maxAnisotropy = 1.0f; + check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui"); + + VkDescriptorSetLayoutBinding tex_bindings[2]{}; + tex_bindings[0].binding = 0; + tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + tex_bindings[0].descriptorCount = 1; + tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; + tex_bindings[1].binding = 1; + tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + tex_bindings[1].descriptorCount = 1; + tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; + VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; + layout_info.bindingCount = 2; + layout_info.pBindings = tex_bindings; + check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex"); + + VkDescriptorSetLayoutBinding bone_bindings[2]{}; + bone_bindings[0].binding = 0; + bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + bone_bindings[0].descriptorCount = 1; + bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT; + bone_bindings[1].binding = 1; + bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + bone_bindings[1].descriptorCount = 1; + bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; + VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; + bone_layout_info.bindingCount = 2; + bone_layout_info.pBindings = bone_bindings; + check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone"); + + VkDescriptorPoolSize pool_sizes[3] = { + {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4 * kMaxFramesInFlight}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4 * kMaxFramesInFlight}}; + VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; + pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; + pool_info.maxSets = 8192; + pool_info.poolSizeCount = 3; + pool_info.pPoolSizes = pool_sizes; + check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool"); + + const std::uint8_t white_pixel[4] = {255, 255, 255, 255}; + upload_texture(1, 1, white_pixel, fallback_texture_, false, false); + fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); + fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); + + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + const auto alignment = std::max( + 1, properties.limits.minStorageBufferOffsetAlignment); + state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; + staging_alignment_ = std::max(16, properties.limits.optimalBufferCopyOffsetAlignment); + + for (auto& slot : frames_) { + void* bone_raw = nullptr; + create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, slot.bone_buffer, slot.bone_memory, &bone_raw); + slot.bone_mapped = static_cast(bone_raw); + slot.bone_capacity = kMaxBonesPerFrame; + for (std::size_t i = 0; i < 16; ++i) slot.bone_mapped[i] = kIdentityMatrix[i]; + + slot.state_capacity = 256; + void* state_raw = nullptr; + create_host_buffer(slot.state_capacity * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + slot.state_buffer, slot.state_memory, &state_raw); + slot.state_mapped = static_cast(state_raw); + + VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + bone_alloc.descriptorPool = descriptor_pool_; + bone_alloc.descriptorSetCount = 1; + bone_alloc.pSetLayouts = &bone_descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &bone_alloc, &slot.bone_descriptor_set), "vkAllocateDescriptorSets bone"); + + VkDescriptorBufferInfo buffer_infos[2] = { + {slot.bone_buffer, 0, kBoneBufferBytes}, + {slot.state_buffer, 0, sizeof(FixedFunctionState)}}; + VkWriteDescriptorSet writes[2]{}; + for (int binding = 0; binding < 2; ++binding) { + writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[binding].dstSet = slot.bone_descriptor_set; + writes[binding].dstBinding = static_cast(binding); + writes[binding].descriptorCount = 1; + writes[binding].descriptorType = binding == 0 + ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + writes[binding].pBufferInfo = &buffer_infos[binding]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + + void* ui_raw = nullptr; + create_host_buffer( + kUiBufferBytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, + slot.ui_buffer, + slot.ui_memory, + &ui_raw); + slot.ui_mapped = static_cast(ui_raw); + slot.ui_vertex_capacity = kMaxUiVerticesPerFrame; + slot.ui_index_capacity = kMaxUiIndicesPerFrame; + slot.ui_vertex_bytes = kUiVertexBytes; + slot.staging_wanted = kStagingRingBytes; + } +} + +void VulkanWindow::release_retired_buffers(FrameSlot& slot) { + for (auto [buffer, memory] : slot.retired_buffers) { + vkDestroyBuffer(device_, buffer, nullptr); + vkFreeMemory(device_, memory, nullptr); + } + slot.retired_buffers.clear(); +} + +void VulkanWindow::reset_staging(FrameSlot& slot) { + release_retired_buffers(slot); + slot.staging_used = 0; + if (slot.staging_wanted <= slot.staging_capacity) return; + if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); + if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); + if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); + void* raw = nullptr; + slot.staging_capacity = std::min(slot.staging_wanted, kStagingRingMaxBytes); + create_host_buffer(slot.staging_capacity, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, + slot.staging_buffer, slot.staging_memory, &raw); + slot.staging_mapped = static_cast(raw); + slot.staging_wanted = slot.staging_capacity; +} + +void VulkanWindow::ensure_bone_capacity(std::size_t required) { + if (required <= f_->bone_capacity) return; + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); + if (required > max_bones || required > UINT32_MAX) + throw std::runtime_error("bone palette exceeds device storage-buffer limit"); + std::size_t next = std::min(max_bones, std::max(required, f_->bone_capacity * 2)); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); + vkUnmapMemory(device_, f_->bone_memory); + vkDestroyBuffer(device_, f_->bone_buffer, nullptr); + vkFreeMemory(device_, f_->bone_memory, nullptr); + f_->bone_buffer = buffer; + f_->bone_memory = memory; + f_->bone_mapped = static_cast(mapped); + f_->bone_capacity = next; + VkDescriptorBufferInfo info{f_->bone_buffer, 0, next * 16 * sizeof(float)}; + VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; + write.dstSet = f_->bone_descriptor_set; + write.dstBinding = 0; + write.descriptorCount = 1; + write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + write.pBufferInfo = &info; + vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); +} + +void VulkanWindow::ensure_state_capacity(std::size_t required) { + if (required <= f_->state_capacity) return; + const auto next = std::max(required, f_->state_capacity * 2); + if (next > UINT32_MAX / state_stride_) + throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + buffer, memory, &mapped); + vkUnmapMemory(device_, f_->state_memory); + vkDestroyBuffer(device_, f_->state_buffer, nullptr); + vkFreeMemory(device_, f_->state_memory, nullptr); + f_->state_buffer = buffer; + f_->state_memory = memory; + f_->state_mapped = static_cast(mapped); + f_->state_capacity = next; + VkDescriptorBufferInfo info{f_->state_buffer, 0, sizeof(FixedFunctionState)}; + VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; + write.dstSet = f_->bone_descriptor_set; + write.dstBinding = 1; + write.descriptorCount = 1; + write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + write.pBufferInfo = &info; + vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); +} + +void VulkanWindow::ensure_ui_capacity(std::size_t command_count) { + if (command_count <= f_->ui_vertex_capacity / 4 && command_count <= f_->ui_index_capacity / 6) return; + if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); + const std::size_t quads = std::max(command_count, f_->ui_vertex_capacity / 2); + const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); + const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); + if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) + throw std::runtime_error("UI geometry exceeds 128 MiB safety limit"); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(vertex_bytes + index_bytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); + vkUnmapMemory(device_, f_->ui_memory); + vkDestroyBuffer(device_, f_->ui_buffer, nullptr); + vkFreeMemory(device_, f_->ui_memory, nullptr); + f_->ui_buffer = buffer; + f_->ui_memory = memory; + f_->ui_mapped = static_cast(mapped); + f_->ui_vertex_capacity = quads * 4; + f_->ui_index_capacity = quads * 6; + f_->ui_vertex_bytes = vertex_bytes; +} + +void VulkanWindow::create_render_pass_and_layout() { + const bool msaa = samples_ != VK_SAMPLE_COUNT_1_BIT; + VkAttachmentDescription attachments[3]{}; + attachments[0].format = swapchain_format_; attachments[0].samples = samples_; + attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; + attachments[0].storeOp = msaa ? VK_ATTACHMENT_STORE_OP_DONT_CARE : VK_ATTACHMENT_STORE_OP_STORE; + attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + attachments[0].finalLayout = msaa ? VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL : VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + // MSAA resolve target: the swapchain image. + attachments[2].format = swapchain_format_; attachments[2].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[2].loadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; attachments[2].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[2].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[2].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = samples_; + attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + + VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; + VkAttachmentReference resolve_ref{2, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkSubpassDescription subpass{}; + subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; + subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; + if (msaa) subpass.pResolveAttachments = &resolve_ref; + subpass.pDepthStencilAttachment = &depth_ref; + + VkSubpassDependency dependency{}; + dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0; + dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; + dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; + dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; + + VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; + pass_info.attachmentCount = msaa ? 3 : 2; pass_info.pAttachments = attachments; + pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; + pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency; + check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass"); + create_framebuffers(); + + const auto vert = read_spirv("native.vert.spv"); + const auto frag = read_spirv("native.frag.spv"); + auto module = [&](const std::vector& code) { + VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO}; + info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data(); + VkShaderModule handle = VK_NULL_HANDLE; + check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule"); + return handle; + }; + vertex_module_ = module(vert); + fragment_module_ = module(frag); + + VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_}; + VkPushConstantRange push_constants{}; + push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; + push_constants.size = sizeof(PushConstants); + VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO}; + layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts; + layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants; + check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); + + get_pipeline(0, 2, 0); + create_offscreen_pass(); +} + +void VulkanWindow::create_offscreen_pass() { + VkFormatProperties properties{}; + vkGetPhysicalDeviceFormatProperties(physical_, VK_FORMAT_R5G6B5_UNORM_PACK16, &properties); + const VkFormatFeatureFlags wanted = + VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT; + offscreen_format_ = (properties.optimalTilingFeatures & wanted) == wanted + ? VK_FORMAT_R5G6B5_UNORM_PACK16 : VK_FORMAT_R8G8B8A8_UNORM; + VkAttachmentDescription attachments[2]{}; + attachments[0].format = offscreen_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[0].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; + attachments[0].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[0].initialLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + attachments[0].finalLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[1].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; + attachments[1].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[1].initialLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; + VkSubpassDescription subpass{}; + subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; + subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; + subpass.pDepthStencilAttachment = &depth_ref; + VkSubpassDependency dependencies[2]{}; + // Earlier work on the queue -- the previous frame sampling the texture, the previous + // offscreen pass, the first-use clear -- completes before this pass writes it. + dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL; dependencies[0].dstSubpass = 0; + dependencies[0].srcStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | + VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | + VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT; + dependencies[0].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | + VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT; + dependencies[0].dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | + VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT; + dependencies[0].dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | + VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; + // The back-buffer pass samples the result. + dependencies[1].srcSubpass = 0; dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL; + dependencies[1].srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; + dependencies[1].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; + dependencies[1].dstStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT; + dependencies[1].dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; + pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; + pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; + pass_info.dependencyCount = 2; pass_info.pDependencies = dependencies; + check(vkCreateRenderPass(device_, &pass_info, nullptr, &offscreen_pass_), "vkCreateRenderPass offscreen"); +} + +VulkanWindow::RenderTarget* VulkanWindow::get_render_target(const std::string& name) { + if (auto it = render_targets_.find(name); it != render_targets_.end()) return &it->second; + unsigned id = 0, w = 0, h = 0; + if (std::sscanf(name.c_str(), "rt:%u:%ux%u", &id, &w, &h) != 3 || !w || !h || w > 4096 || h > 4096) + return nullptr; + RenderTarget& target = render_targets_[name]; + target.width = w; + target.height = h; + auto make_image = [&](VkFormat format, VkImageUsageFlags usage, VkImageAspectFlags aspect, + VkImage& image, VkDeviceMemory& memory, VkImageView& view) { + VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + info.imageType = VK_IMAGE_TYPE_2D; + info.format = format; + info.extent = {w, h, 1}; + info.mipLevels = 1; info.arrayLayers = 1; + info.samples = VK_SAMPLE_COUNT_1_BIT; + info.tiling = VK_IMAGE_TILING_OPTIMAL; + info.usage = usage; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &info, nullptr, &image), "vkCreateImage render target"); + VkMemoryRequirements reqs{}; + vkGetImageMemoryRequirements(device_, image, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory render target"); + check(vkBindImageMemory(device_, image, memory, 0), "vkBindImageMemory render target"); + VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view_info.image = image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = format; + view_info.subresourceRange = {aspect, 0, 1, 0, 1}; + check(vkCreateImageView(device_, &view_info, nullptr, &view), "vkCreateImageView render target"); + }; + make_image(offscreen_format_, + VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT, + VK_IMAGE_ASPECT_COLOR_BIT, target.texture.image, target.texture.memory, target.texture.view); + make_image(VK_FORMAT_D32_SFLOAT, + VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, + VK_IMAGE_ASPECT_DEPTH_BIT, target.depth_image, target.depth_memory, target.depth_view); + const VkImageView views[2] = {target.texture.view, target.depth_view}; + VkFramebufferCreateInfo fb{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + fb.renderPass = offscreen_pass_; + fb.attachmentCount = 2; fb.pAttachments = views; + fb.width = w; fb.height = h; fb.layers = 1; + check(vkCreateFramebuffer(device_, &fb, nullptr, &target.framebuffer), "vkCreateFramebuffer render target"); + target.texture.descriptor = allocate_texture_descriptor_set(target.texture.view, fallback_texture_.view); + target.texture.ui_descriptor = allocate_ui_texture_descriptor_set(target.texture.view, fallback_texture_.view); + return ⌖ +} + +void VulkanWindow::release_render_target(RenderTarget& target) { + if (target.framebuffer) vkDestroyFramebuffer(device_, target.framebuffer, nullptr); + if (target.depth_view) vkDestroyImageView(device_, target.depth_view, nullptr); + if (target.depth_image) vkDestroyImage(device_, target.depth_image, nullptr); + if (target.depth_memory) vkFreeMemory(device_, target.depth_memory, nullptr); + release_texture(target.texture); + target = {}; +} + +void VulkanWindow::initialize_render_target(RenderTarget& target) { + VkImageMemoryBarrier barriers[2]{}; + for (auto& b : barriers) { + b.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER; + b.srcQueueFamilyIndex = b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + b.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + b.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + b.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + } + barriers[0].image = target.texture.image; + barriers[0].subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + barriers[1].image = target.depth_image; + barriers[1].subresourceRange = {VK_IMAGE_ASPECT_DEPTH_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 2, barriers); + const VkClearColorValue white{{1.0f, 1.0f, 1.0f, 1.0f}}; + const VkClearDepthStencilValue far{1.0f, 0}; + vkCmdClearColorImage(f_->command, target.texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &white, 1, + &barriers[0].subresourceRange); + vkCmdClearDepthStencilImage(f_->command, target.depth_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &far, 1, + &barriers[1].subresourceRange); + for (auto& b : barriers) { + b.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + b.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + } + barriers[0].newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barriers[0].dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_READ_BIT; + barriers[1].newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + barriers[1].dstAccessMask = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | + VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, + 0, 0, nullptr, 0, nullptr, 2, barriers); + target.initialized = true; +} + +VkPipeline VulkanWindow::get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, + bool lines, std::uint8_t z_func, bool offscreen) { + const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | + (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | + (std::uint32_t(z_func & 0xfu) << 25) | (offscreen ? (1u << 29) : 0u); + if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; + + VkPipelineShaderStageCreateInfo stages[2]{}; + stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; + stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main"; + stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main"; + + VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX}; + VkVertexInputAttributeDescription attributes[9] = { + {0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)}, + {1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)}, + {2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)}, + {3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)}, + {4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)}, + {5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)}, + {6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)}, + {7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)}, + {8, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, vertex_fog)}}; + VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO}; + input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding; + input.vertexAttributeDescriptionCount = 9; input.pVertexAttributeDescriptions = attributes; + + VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO}; + assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; + + VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; + // An offscreen target is at most the device's 4096 (GetDeviceCaps); the viewport clips inside it. + VkRect2D scissor{{0, 0}, offscreen ? VkExtent2D{4096, 4096} : extent_}; + VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; + viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; + viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; + // D3DVIEWPORT8 changes per draw (SetViewport), so the viewport is dynamic state. + const VkDynamicState dynamic_states[] = {VK_DYNAMIC_STATE_VIEWPORT}; + VkPipelineDynamicStateCreateInfo dynamic_info{VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO}; + dynamic_info.dynamicStateCount = 1; dynamic_info.pDynamicStates = dynamic_states; + + VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO}; + raster.polygonMode = VK_POLYGON_MODE_FILL; + raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE); + raster.frontFace = VK_FRONT_FACE_CLOCKWISE; + raster.lineWidth = 1; + + VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; + multisample.rasterizationSamples = offscreen ? VK_SAMPLE_COUNT_1_BIT : samples_; + + VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; + depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; + depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE; + switch (z_func) { + case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break; + case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break; + case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break; + case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break; + case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break; + case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break; + case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break; + default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break; + } + + auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor { + switch (d3d) { + case 1: return VK_BLEND_FACTOR_ZERO; + case 2: return VK_BLEND_FACTOR_ONE; + case 3: return VK_BLEND_FACTOR_SRC_COLOR; + case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR; + case 5: return VK_BLEND_FACTOR_SRC_ALPHA; + case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + case 7: return VK_BLEND_FACTOR_DST_ALPHA; + case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; + case 9: return VK_BLEND_FACTOR_DST_COLOR; + case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR; + case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE; + default: return fallback; + } + }; + + VkPipelineColorBlendAttachmentState blend_attachment{}; + blend_attachment.colorWriteMask = 0xf; + if (blend > 0) { + blend_attachment.blendEnable = VK_TRUE; + if (blend >= 16) { + blend_attachment.srcColorBlendFactor = + d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA); + blend_attachment.dstColorBlendFactor = + d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA); + } else { + blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA; + blend_attachment.dstColorBlendFactor = + blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + } + blend_attachment.colorBlendOp = VK_BLEND_OP_ADD; + // D3D8 has no separate alpha blend: SRCBLEND/DESTBLEND apply to alpha as well. + auto alpha_factor = [](VkBlendFactor factor) { + switch (factor) { + case VK_BLEND_FACTOR_SRC_COLOR: return VK_BLEND_FACTOR_SRC_ALPHA; + case VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + case VK_BLEND_FACTOR_DST_COLOR: return VK_BLEND_FACTOR_DST_ALPHA; + case VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; + case VK_BLEND_FACTOR_SRC_ALPHA_SATURATE: return VK_BLEND_FACTOR_ONE; + default: return factor; + } + }; + blend_attachment.srcAlphaBlendFactor = alpha_factor(blend_attachment.srcColorBlendFactor); + blend_attachment.dstAlphaBlendFactor = alpha_factor(blend_attachment.dstColorBlendFactor); + blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD; + } + VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO}; + blending.attachmentCount = 1; blending.pAttachments = &blend_attachment; + + VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO}; + pipeline_info.stageCount = 2; pipeline_info.pStages = stages; + pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly; + pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster; + pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; + pipeline_info.pColorBlendState = &blending; + pipeline_info.pDynamicState = &dynamic_info; + pipeline_info.layout = layout_; pipeline_info.renderPass = offscreen ? offscreen_pass_ : pass_; + VkPipeline handle = VK_NULL_HANDLE; + check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); + pipelines_.emplace(key, handle); + return handle; +} + +VkDescriptorSet VulkanWindow::allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, + VkSampler sampler0, + VkSampler sampler1) { + VkDescriptorSet set = VK_NULL_HANDLE; + VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + desc_alloc.descriptorPool = descriptor_pool_; + desc_alloc.descriptorSetCount = 1; + desc_alloc.pSetLayouts = &descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets"); + + VkDescriptorImageInfo image_infos[2]{}; + image_infos[0].sampler = sampler0 ? sampler0 : sampler_; + image_infos[0].imageView = view0; + image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_); + image_infos[1].imageView = view1; + image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + + VkWriteDescriptorSet writes[2]{}; + for (int b = 0; b < 2; ++b) { + writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[b].dstSet = set; + writes[b].dstBinding = static_cast(b); + writes[b].descriptorCount = 1; + writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + writes[b].pImageInfo = &image_infos[b]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + return set; +} + +VkDescriptorSet VulkanWindow::allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) { + VkDescriptorSet set = VK_NULL_HANDLE; + VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + desc_alloc.descriptorPool = descriptor_pool_; + desc_alloc.descriptorSetCount = 1; + desc_alloc.pSetLayouts = &descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui"); + + VkDescriptorImageInfo image_infos[2]{}; + image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_; + image_infos[0].imageView = view0; + image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_); + image_infos[1].imageView = view1; + image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + + VkWriteDescriptorSet writes[2]{}; + for (int b = 0; b < 2; ++b) { + writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[b].dstSet = set; + writes[b].dstBinding = static_cast(b); + writes[b].descriptorCount = 1; + writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + writes[b].pImageInfo = &image_infos[b]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + return set; +} + +VulkanWindow::GpuTexture& VulkanWindow::get_gpu_texture( + const std::string& texture_name, + const std::unordered_map>& capture_textures) { + if (texture_name.empty()) return fallback_texture_; + if (texture_name.rfind("rt:", 0) == 0) { + RenderTarget* target = get_render_target(texture_name); + return target ? target->texture : fallback_texture_; + } + auto [it, inserted] = textures_.try_emplace(texture_name); + it->second.last_used_frame = frame_number_; + if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) + return it->second; + if (inserted) { + it->second.descriptor = fallback_texture_.descriptor; + it->second.ui_descriptor = fallback_texture_.ui_descriptor; + } + it->second.last_decode_attempt_frame = frame_number_; + + mtimage::Image decoded; + if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) { + decoded = decode_texture_bytes(raw->second.data(), raw->second.size()); + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (!decoded.ok()) { + if (texture_name.rfind("mem:", 0) == 0) { + UIMemoryTexture mem_tex; + if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { + const auto mtra = native_draw_capture::encode_raw_argb_as_mtra( + static_cast(mem_tex.width), + static_cast(mem_tex.height), + mem_tex.argb.data()); + decoded = decode_texture_bytes(mtra.data(), mtra.size()); + } + } else { + std::vector pack_bytes; + if (read_live_pack_texture(texture_name, pack_bytes)) + decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size()); + } + } +#endif + if (!decoded.ok()) return it->second; + // 40250 CGraphicImageTexture: DXT DDS -> CreateDDSTexture with the file's levels; + // other files -> D3DXCreateTextureFromFileInMemoryEx(D3DX_DEFAULT) = full chain; + // memory images (mem:, CreateTexture(w, h, 1, ...)) -> one level. + const bool is_memory_tex = texture_name.rfind("mem:", 0) == 0; + last_texture_upload_name_ = texture_name; + upload_texture( + decoded.w, decoded.h, decoded.rgba.data(), it->second, true, + !is_memory_tex && decoded.file_mip_levels == 0, &decoded.mips); + it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view); + it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view); + return it->second; +} + +std::uint64_t VulkanWindow::sampler_state_key(const Render3DDraw& draw, int stage) const { + return (std::uint64_t(draw.address_u[stage] & 15u)) | + (std::uint64_t(draw.address_v[stage] & 15u) << 4) | + (std::uint64_t(draw.min_filter[stage] & 15u) << 8) | + (std::uint64_t(draw.mag_filter[stage] & 15u) << 12) | + (std::uint64_t(draw.mip_filter[stage] & 15u) << 16); +} + +VkSampler VulkanWindow::sampler_for_draw(const Render3DDraw& draw, int stage) { + const auto key = sampler_state_key(draw, stage); + if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_; + if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second; + auto address_mode = [](std::uint32_t mode) { + switch (mode) { + case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT; + case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER; + default: return VK_SAMPLER_ADDRESS_MODE_REPEAT; + } + }; + VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + info.addressModeU = address_mode(draw.address_u[stage]); + info.addressModeV = address_mode(draw.address_v[stage]); + info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; + info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE; + info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; + info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; + info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR; + info.minLod = 0.0f; + info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE; + if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) { + info.anisotropyEnable = VK_TRUE; + info.maxAnisotropy = max_anisotropy_; + } + VkSampler sampler = VK_NULL_HANDLE; + check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state"); + state_samplers_.emplace(key, sampler); + return sampler; +} + +VkDescriptorSet VulkanWindow::get_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures, + const Render3DDraw& draw) { + const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); + const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures); + const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" + + std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1)); + if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { + it->second.last_used_frame = frame_number_; + return it->second.set; + } + const VkDescriptorSet set = allocate_texture_descriptor_set( + tex0.view ? tex0.view : fallback_texture_.view, + tex1.view ? tex1.view : fallback_texture_.view, + sampler_for_draw(draw, 0), sampler_for_draw(draw, 1)); + paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); + return set; +} + +VkDescriptorSet VulkanWindow::get_ui_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures) { + const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); + if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; + const auto& tex1 = get_gpu_texture(mask_name, capture_textures); + if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; + const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name; + if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { + it->second.last_used_frame = frame_number_; + return it->second.set; + } + const VkDescriptorSet set = allocate_ui_texture_descriptor_set( + tex0.view ? tex0.view : fallback_texture_.view, + tex1.view ? tex1.view : fallback_texture_.view); + paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); + return set; +} + +void VulkanWindow::upload_texture( + std::uint32_t width, + std::uint32_t height, + const std::uint8_t* rgba, + GpuTexture& texture, + bool count_upload, + bool generate_mips, + const std::vector>* file_mips) { + const ScopedMs timer(timings_.texture_upload_ms); + const bool has_file_mips = !generate_mips && file_mips && !file_mips->empty(); + VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; + if (has_file_mips) + for (const auto& level : *file_mips) byte_size += level.size(); + const std::uint32_t mip_levels = generate_mips + ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) + : has_file_mips ? 1u + static_cast(file_mips->size()) : 1u; + // Inside a frame the copy is recorded ahead of that frame's render pass and reads the slot's + // staging ring; the GPU runs it with the frame, no queue wait. Outside a frame (the fallback + // texture at startup) it is a one-off submit that waits. + const bool in_frame = f_->recording; + VkBuffer staging_buffer = VK_NULL_HANDLE; + VkDeviceMemory staging_memory = VK_NULL_HANDLE; + VkDeviceSize staging_offset = 0; + std::uint8_t* mapped = nullptr; + const VkDeviceSize aligned_offset = + (f_->staging_used + staging_alignment_ - 1) / staging_alignment_ * staging_alignment_; + if (in_frame && f_->staging_mapped && aligned_offset + byte_size <= f_->staging_capacity) { + staging_buffer = f_->staging_buffer; + staging_offset = aligned_offset; + mapped = f_->staging_mapped + staging_offset; + f_->staging_used = aligned_offset + byte_size; + } else { + if (in_frame) f_->staging_wanted = std::max(f_->staging_wanted, aligned_offset + byte_size); + void* raw = nullptr; + create_host_buffer(byte_size, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, staging_buffer, staging_memory, &raw); + mapped = static_cast(raw); + } + std::memcpy(mapped, rgba, VkDeviceSize(width) * height * 4); + if (has_file_mips) { + auto* dst = mapped + VkDeviceSize(width) * height * 4; + for (const auto& level : *file_mips) { std::memcpy(dst, level.data(), level.size()); dst += level.size(); } + } + if (staging_memory) vkUnmapMemory(device_, staging_memory); + + VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + image_info.imageType = VK_IMAGE_TYPE_2D; + image_info.format = VK_FORMAT_R8G8B8A8_UNORM; + image_info.extent = {width, height, 1}; + image_info.mipLevels = mip_levels; + image_info.arrayLayers = 1; + image_info.samples = VK_SAMPLE_COUNT_1_BIT; + image_info.tiling = VK_IMAGE_TILING_OPTIMAL; + image_info.usage = + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT; + image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture"); + VkMemoryRequirements image_reqs{}; + vkGetImageMemoryRequirements(device_, texture.image, &image_reqs); + VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + image_alloc.allocationSize = image_reqs.size; + image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); + check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); + + VkCommandBuffer upload_cmd = f_->command; + if (!in_frame) { + VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); + } + + VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_transfer.image = texture.image; + to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &to_transfer); + + std::vector regions(has_file_mips ? mip_levels : 1u); + VkDeviceSize region_offset = 0; + for (std::uint32_t i = 0; i < regions.size(); ++i) { + const std::uint32_t lw = std::max(1u, width >> i), lh = std::max(1u, height >> i); + regions[i].bufferOffset = staging_offset + region_offset; + regions[i].imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; + regions[i].imageExtent = {lw, lh, 1}; + region_offset += VkDeviceSize(lw) * lh * 4; + } + vkCmdCopyBufferToImage( + upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + static_cast(regions.size()), regions.data()); + + std::int32_t mip_w = static_cast(width); + std::int32_t mip_h = static_cast(height); + for (std::uint32_t i = 1; generate_mips && i < mip_levels; ++i) { + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.image = texture.image; + barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1}; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &barrier); + + const std::int32_t next_w = std::max(1, mip_w / 2); + const std::int32_t next_h = std::max(1, mip_h / 2); + VkImageBlit blit{}; + blit.srcOffsets[0] = {0, 0, 0}; + blit.srcOffsets[1] = {mip_w, mip_h, 1}; + blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1}; + blit.dstOffsets[0] = {0, 0, 0}; + blit.dstOffsets[1] = {next_w, next_h, 1}; + blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; + vkCmdBlitImage( + upload_cmd, + texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + 1, &blit, VK_FILTER_LINEAR); + + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &barrier); + + mip_w = next_w; + mip_h = next_h; + } + + VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_shader.image = texture.image; + to_shader.subresourceRange = generate_mips + ? VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1} + : VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &to_shader); + + if (in_frame) { + if (staging_memory) f_->retired_buffers.emplace_back(staging_buffer, staging_memory); + } else { + check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; + check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); + check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); + vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); + vkDestroyBuffer(device_, staging_buffer, nullptr); + vkFreeMemory(device_, staging_memory, nullptr); + } + + VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; + view_info.format = VK_FORMAT_R8G8B8A8_UNORM; + view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture"); + + if (count_upload) { + ++texture_upload_count_; + texture_uploaded_bytes_ += byte_size; + } +} + +void VulkanWindow::release_texture(GpuTexture& texture) { + for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor}) + if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor) + vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set); + if (texture.view) vkDestroyImageView(device_, texture.view, nullptr); + if (texture.image) vkDestroyImage(device_, texture.image, nullptr); + if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr); + texture = {}; +} + +void VulkanWindow::prune_textures() { + constexpr std::uint64_t static_retention_frames = 120; + constexpr std::uint64_t memory_retention_frames = 2; + for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) { + const auto retention = it->first.find("mem:") != std::string::npos + ? memory_retention_frames : static_retention_frames; + if (frame_number_ - it->second.last_used_frame > retention) { + vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set); + it = paired_descriptors_.erase(it); + } else ++it; + } + for (auto it = textures_.begin(); it != textures_.end();) { + const auto retention = it->first.rfind("mem:", 0) == 0 + ? memory_retention_frames : static_retention_frames; + if (frame_number_ - it->second.last_used_frame > retention) { + release_texture(it->second); + it = textures_.erase(it); + } else ++it; + } +} + +void VulkanWindow::release_geometry(Geometry& geometry) { + if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr); + if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr); + geometry = {}; +} + +void VulkanWindow::upload_geometry(const Render3DDraw& draw, Geometry& geometry) { + const ScopedMs timer(timings_.geometry_upload_ms); + if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) + throw std::runtime_error("invalid native geometry array size"); + const auto count = draw.positions.size() / 3; + if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX) + throw std::runtime_error("geometry exceeds Vulkan index range"); + // D3D8 feeds opaque white for a vertex format without D3DFVF_DIFFUSE. + const std::uint32_t default_argb = 0xffffffffu; + const bool has_skin = + draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4; + std::vector vertices(count); + for (std::size_t i = 0; i < count; ++i) { + vertices[i].position[0] = draw.positions[3*i]; + vertices[i].position[1] = draw.positions[3*i+1]; + vertices[i].position[2] = draw.positions[3*i+2]; + vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f; + vertices[i].vertex_fog = i < draw.vertex_fog.size() ? draw.vertex_fog[i] : 1.0f; + if (3*i + 2 < draw.normals.size()) { + vertices[i].normal[0] = draw.normals[3*i]; + vertices[i].normal[1] = draw.normals[3*i+1]; + vertices[i].normal[2] = draw.normals[3*i+2]; + } else { + vertices[i].normal[0] = 0.0f; + vertices[i].normal[1] = 0.0f; + vertices[i].normal[2] = 1.0f; + } + if (2*i + 1 < draw.uv0.size()) { + vertices[i].uv[0] = draw.uv0[2*i]; + vertices[i].uv[1] = draw.uv0[2*i+1]; + } else { + vertices[i].uv[0] = 0.0f; + vertices[i].uv[1] = 0.0f; + } + const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb; + vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f; + vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f; + vertices[i].color[2] = float(argb & 255) / 255.0f; + vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f; + if (has_skin) { + for (int k = 0; k < 4; ++k) { + vertices[i].joints[k] = draw.bone_indices[4*i + k]; + vertices[i].weights[k] = draw.bone_weights[4*i + k]; + } + } else { + vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0; + vertices[i].weights[0] = 1.0f; + vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f; + } + if (2*i + 1 < draw.uv1.size()) { + vertices[i].mask_uv[0] = draw.uv1[2*i]; + vertices[i].mask_uv[1] = draw.uv1[2*i+1]; + } else { + vertices[i].mask_uv[0] = 0.0f; + vertices[i].mask_uv[1] = 0.0f; + } + } + for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count"); + geometry.index_offset = vertices.size() * sizeof(Vertex); + const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t); + if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB"); + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type( + requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory"); + check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory"); + void* mapped = nullptr; + check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory"); + std::memcpy(mapped, vertices.data(), geometry.index_offset); + std::memcpy(static_cast(mapped) + geometry.index_offset, + draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); + vkUnmapMemory(device_, geometry.memory); + geometry.vertex_count = static_cast(count); + geometry.index_count = static_cast(draw.indices.size()); + ++upload_count_; + uploaded_bytes_ += bytes; +} + +void VulkanWindow::update_fps_counter() { + const auto now = std::chrono::steady_clock::now(); + if (fps_frames_++ == 0) { + fps_window_start_ = now; + return; + } + const double elapsed = std::chrono::duration(now - fps_window_start_).count(); + if (elapsed >= 0.5) { + fps_ = int(std::lround((fps_frames_ - 1) / elapsed)); + fps_frames_ = 1; + fps_window_start_ = now; + } +} + +void VulkanWindow::note_input(const SDL_Event& event) { + switch (event.type) { + case SDL_EVENT_FINGER_DOWN: ++fingers_down_; break; + case SDL_EVENT_FINGER_UP: + case SDL_EVENT_FINGER_CANCELED: fingers_down_ = std::max(0, fingers_down_ - 1); break; + case SDL_EVENT_MOUSE_BUTTON_DOWN: ++buttons_down_; break; + case SDL_EVENT_MOUSE_BUTTON_UP: buttons_down_ = std::max(0, buttons_down_ - 1); break; + case SDL_EVENT_WINDOW_FOCUS_LOST: case SDL_EVENT_WILL_ENTER_BACKGROUND: + // An up event lost with the focus must not keep the client out of the idle rate for good. + fingers_down_ = buttons_down_ = 0; + break; + case SDL_EVENT_FINGER_MOTION: case SDL_EVENT_MOUSE_MOTION: case SDL_EVENT_MOUSE_WHEEL: + case SDL_EVENT_KEY_DOWN: case SDL_EVENT_KEY_UP: case SDL_EVENT_TEXT_INPUT: + case SDL_EVENT_WINDOW_FOCUS_GAINED: case SDL_EVENT_DID_ENTER_FOREGROUND: break; + default: return; + } + last_input_ = std::chrono::steady_clock::now(); +} + +void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms, + std::vector frame_ms) { + const auto timing = renderer.timings(); + std::sort(frame_ms.begin(), frame_ms.end()); + const auto percentile = [&](double fraction) { + if (frame_ms.empty()) return 0.0; + const auto index = static_cast(std::ceil(fraction * frame_ms.size())) - 1; + return frame_ms[std::min(index, frame_ms.size() - 1)]; + }; + std::cout << "device=" << renderer.device_name() + << " present_mode=" << renderer.present_mode_name() << " msaa=" << renderer.msaa_samples() + << " frames=" << completed + << " draws=" << renderer.draw_count() + << " skinned_draws=" << renderer.skinned_draw_count() + << " ui_batches=" << renderer.ui_batch_count() + << " ui_quads=" << renderer.ui_quad_count() + << " vertices=" << renderer.vertex_count() + << " indices=" << renderer.index_count() + << " uploads=" << renderer.upload_count() + << " uploaded_bytes=" << renderer.uploaded_bytes() + << " texture_uploads=" << renderer.texture_upload_count() + << " texture_bytes=" << renderer.texture_uploaded_bytes() + << " mean_frame_ms=" << (completed ? wall_ms / completed : 0.0) + << " p95_frame_ms=" << percentile(0.95) + << " p99_frame_ms=" << percentile(0.99) + << " max_frame_ms=" << percentile(1.0) + << " game_update_ms=" << (completed ? update_ms / completed : 0.0) + << " sync_ms=" << (completed ? timing.sync_ms / completed : 0.0) + << " prepare_ms=" << (completed ? timing.prepare_ms / completed : 0.0) + << " steady_prepare_ms=" << (completed > 1 ? timing.steady_prepare_ms / (completed - 1) : 0.0) + << " submit_ms=" << (completed ? timing.submit_ms / completed : 0.0) + << " present_ms=" << (completed ? timing.present_ms / completed : 0.0) + << " gpu_ms=" << (timing.gpu_samples ? timing.gpu_ms / double(timing.gpu_samples) : 0.0) << '\n'; +} + +} // namespace mt_host diff --git a/src/host/vulkan_window.h b/src/host/vulkan_window.h new file mode 100644 index 00000000..138353cc --- /dev/null +++ b/src/host/vulkan_window.h @@ -0,0 +1,437 @@ +#pragma once + +#include "RenderCommands3D.h" +#include "UIRenderCommands.h" +#include "perf_log.h" +#include "render_state.h" +#include "touch_controller.h" + +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace mt_host { + +// The SDL3 window, Vulkan device and swapchain, and the renderer that replays a frame's recorded +// D3D8 draws (Render3DDraw) and UI commands. +class VulkanWindow { + struct FrameSlot; + struct GeometryId { + std::uint64_t key = 0, signature = 0; + bool operator==(const GeometryId&) const = default; + }; + struct GeometryIdHash { + std::size_t operator()(GeometryId id) const { + return std::size_t(id.key ^ (id.signature + 0x9e3779b97f4a7c15ull + (id.key << 6) + (id.key >> 2))); + } + }; + struct Geometry { + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + VkDeviceSize index_offset = 0; + std::uint64_t last_used_frame = 0; + std::uint32_t vertex_count = 0, index_count = 0; + }; + struct GpuTexture { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + VkImageView view = VK_NULL_HANDLE; + VkDescriptorSet descriptor = VK_NULL_HANDLE; + VkDescriptorSet ui_descriptor = VK_NULL_HANDLE; + std::uint64_t last_used_frame = 0; + std::uint64_t last_decode_attempt_frame = 0; + }; + struct PairedDescriptor { + VkDescriptorSet set = VK_NULL_HANDLE; + std::uint64_t last_used_frame = 0; + }; + // An offscreen D3D render-target texture (the 40250 character shadow map, "rt::x"): + // drawn by the frame's offscreen pass ahead of the back-buffer pass and sampled by that pass. + // Between frames the colour image stays SHADER_READ_ONLY_OPTIMAL and the depth image + // DEPTH_STENCIL_ATTACHMENT_OPTIMAL; the render pass keeps (LOAD/STORE) both, like a D3D surface. + struct RenderTarget { + GpuTexture texture; // colour image, view and descriptors used for sampling + VkImage depth_image = VK_NULL_HANDLE; + VkDeviceMemory depth_memory = VK_NULL_HANDLE; + VkImageView depth_view = VK_NULL_HANDLE; + VkFramebuffer framebuffer = VK_NULL_HANDLE; + std::uint32_t width = 0, height = 0; + bool initialized = false; // cleared to white / depth 1 by the first frame that uses it + }; + struct PreparedDraw { + Geometry* geometry = nullptr; + VkPipeline pipeline = VK_NULL_HANDLE; + VkDescriptorSet descriptor = VK_NULL_HANDLE; + PushConstants constants{}; + FixedFunctionState fixed_state{}; + std::uint32_t state_offset = 0; + VkViewport viewport{}; + // Offscreen render target of the draw, or null for the back buffer. + RenderTarget* target = nullptr; + // IDirect3DDevice8::Clear recorded in draw order; geometry is null for these. + std::uint32_t clear_flags = 0; + VkClearValue clear_color{}; + VkClearValue clear_depth{}; + VkRect2D clear_rect{}; + }; + struct UiBatch { + std::uint32_t first_index = 0; + std::uint32_t index_count = 0; + VkPipeline pipeline = VK_NULL_HANDLE; + VkDescriptorSet descriptor = VK_NULL_HANDLE; + PushConstants constants{}; + bool behind_3d = false; + std::uint32_t state_offset = 0; + }; + + static constexpr std::size_t kMaxBonesPerFrame = 65536; + static constexpr VkDeviceSize kBoneBufferBytes = kMaxBonesPerFrame * 16 * sizeof(float); + static constexpr std::size_t kMaxUiVerticesPerFrame = 65536; + static constexpr std::size_t kMaxUiIndicesPerFrame = 98304; + static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex); + static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t); + static constexpr VkDeviceSize kStagingRingBytes = 4ull * 1024 * 1024; + static constexpr VkDeviceSize kStagingRingMaxBytes = 64ull * 1024 * 1024; + +public: + struct Timings { + double sync_ms = 0; + double fence_ms = 0; // inside sync_ms: waiting for the frame slot's previous GPU work + double prepare_ms = 0; + double steady_prepare_ms = 0; + double submit_ms = 0; + double present_ms = 0; + double gpu_ms = 0; + std::uint64_t gpu_samples = 0; + // Inside prepare_ms: creating geometry buffers and the blocking texture uploads. + double geometry_upload_ms = 0; + double texture_upload_ms = 0; + }; + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + void enable_game_hardware_cursor(); + + void sync_game_cursor(); +#endif + + explicit VulkanWindow(bool vsync = true, int init_width = 960, int init_height = 640); + + ~VulkanWindow(); + + // 32-bit bottom-up BMP from a B8G8R8A8 / R8G8B8A8 swapchain readback. + void write_bmp(const std::string& path, const std::uint8_t* pixels) const; + + // Seconds since the last touch, mouse or key event; 0 while a finger or mouse button is held (a held + // joystick or attack button sends no events). The frame-rate policy's idle test. + double input_idle_seconds() const; + + void request_screenshot(std::string path, std::uint64_t frame); + std::uint64_t frame_number() const { return frame_number_; } + + int msaa_samples() const { return int(samples_); } + std::uint32_t width() const { return extent_.width; } + std::uint32_t height() const { return extent_.height; } + std::uint32_t logical_width() const; + std::uint32_t logical_height() const; + + VkViewport full_viewport() const; + + // Window pixels kept clear at the left and right screen edges (rounded corners, cutouts); + // MainActivity measures them and passes --safe-inset-px. + void set_safe_inset_px(int px) { safe_inset_px_ = std::max(0, px); } + // Test builds: frames per second in the top-left corner (--show-fps). + void set_show_fps(bool show) { show_fps_ = show; } + SDL_Window* sdl_window() const { return window_; } + // The last render()'s blocking time: slot fence + swapchain acquire. + double last_sync_ms() const { return last_sync_ms_; } + // The same inset in logical UI pixels: the 40250 window layers are shifted and narrowed by it. + int ui_safe_inset() const; + + // D3DVIEWPORT8 (in the game's logical screen pixels) scaled to the swapchain extent. + VkViewport draw_viewport(const Render3DDraw& draw) const; + + TouchController touch_controller; + + // Touch hosts: SDL text input (and with it the system keyboard) is on only while the player + // types into a tapped EditLine; desktop keeps the always-on text input from window creation. + void sync_screen_keyboard(); + + bool poll(bool forward_to_live_client = false); + + void reset_timings(); + + void finish_gpu_timings(); + + // 1 = the old serial loop (CPU waits for the previous frame's GPU work), 2 = CPU/GPU overlap. + void set_frames_in_flight(std::uint32_t count); + std::uint32_t frames_in_flight() const { return frames_in_flight_; } + + void render( + const std::vector& draws, + const std::unordered_map>& capture_textures, + std::uint32_t ui_width = 960, + std::uint32_t ui_height = 640, + const std::vector& ui_commands = {}); + + std::size_t draw_count() const { return last_draw_count_; } + std::size_t skinned_draw_count() const { return last_skinned_draw_count_; } + std::size_t ui_batch_count() const { return last_ui_batch_count_; } + std::size_t ui_quad_count() const { return last_ui_quad_count_; } + std::size_t vertex_count() const { return last_vertex_count_; } + std::size_t index_count() const { return last_index_count_; } + std::size_t upload_count() const { return upload_count_; } + std::size_t uploaded_bytes() const { return uploaded_bytes_; } + std::size_t texture_upload_count() const { return texture_upload_count_; } + std::size_t texture_uploaded_bytes() const { return texture_uploaded_bytes_; } + const std::string& device_name() const { return device_name_; } + const char* present_mode_name() const; + Timings timings() const { return timings_; } + native_perf::RendererTotals perf_totals() const; + +private: + void recreate_swapchain(); + + void collect_pending_gpu_timestamp(FrameSlot& slot); + + void build_ui_batches( + std::uint32_t ui_width, + std::uint32_t ui_height, + const std::vector& commands, + const std::unordered_map>& capture_textures, + std::vector& batches, + std::size_t& out_quad_count); + + std::uint32_t find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const; + + void select_device(); + + void create_swapchain(); + + // The multisampled colour target the subpass resolves into the swapchain image. It never + // leaves the render pass, so it is transient (lazily allocated tile memory where available). + void create_msaa_color(); + + void destroy_msaa_color(); + + void create_framebuffers(); + + void create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped); + + void create_descriptors_and_buffers(); + + // The slot's fence has signalled: its oversize staging buffers are free, and the ring restarts + // (grown to the largest frame's uploads seen so far, so texture streaming stays in the ring). + void release_retired_buffers(FrameSlot& slot); + + void reset_staging(FrameSlot& slot); + + void ensure_bone_capacity(std::size_t required); + + void ensure_state_capacity(std::size_t required); + + void ensure_ui_capacity(std::size_t command_count); + + void create_render_pass_and_layout(); + + // The render pass of offscreen render-target textures: colour (R5G6B5 like the D3D surface when + // the device can render to it) + D32 depth, both loaded and stored, colour left shader-readable. + void create_offscreen_pass(); + + // "rt::x" -> its render target, created on first use (by a draw into it or by a draw + // sampling it); null for a malformed name. + RenderTarget* get_render_target(const std::string& name); + + void release_render_target(RenderTarget& target); + + // A new render target starts white with depth 1 (what the CPU surface's first Clear leaves), in + // the layouts offscreen_pass_ expects. Recorded outside any render pass. + void initialize_render_target(RenderTarget& target); + + // offscreen: for offscreen_pass_ (single-sampled render-target texture) instead of pass_. + VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, + bool lines = false, std::uint8_t z_func = 4, bool offscreen = false); + + VkDescriptorSet allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, + VkSampler sampler0 = VK_NULL_HANDLE, + VkSampler sampler1 = VK_NULL_HANDLE); + + VkDescriptorSet allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1); + + GpuTexture& get_gpu_texture( + const std::string& texture_name, + const std::unordered_map>& capture_textures); + + std::uint64_t sampler_state_key(const Render3DDraw& draw, int stage) const; + + VkSampler sampler_for_draw(const Render3DDraw& draw, int stage); + + VkDescriptorSet get_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures, + const Render3DDraw& draw); + + VkDescriptorSet get_ui_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures); + + void upload_texture( + std::uint32_t width, + std::uint32_t height, + const std::uint8_t* rgba, + GpuTexture& texture, + bool count_upload, + bool generate_mips = true, + const std::vector>* file_mips = nullptr); + + void release_texture(GpuTexture& texture); + + void prune_textures(); + + void release_geometry(Geometry& geometry); + + void upload_geometry(const Render3DDraw& draw, Geometry& geometry); + + bool vsync_ = true; + VkPresentModeKHR present_mode_ = VK_PRESENT_MODE_FIFO_KHR; + float timestamp_period_ns_ = 1.0f; + float max_anisotropy_ = 1.0f; + SDL_Window* window_ = nullptr; + unsigned screen_keyboard_serial_ = 0; + int safe_inset_px_ = 0; + bool show_fps_ = false; + int fps_ = -1; + int fps_frames_ = 0; + std::chrono::steady_clock::time_point fps_window_start_{}; + + // Adds the scope's wall time to a Timings field. + struct ScopedMs { + explicit ScopedMs(double& into) : into_(into), start_(std::chrono::steady_clock::now()) {} + ~ScopedMs() { into_ += std::chrono::duration(std::chrono::steady_clock::now() - start_).count(); } + double& into_; + std::chrono::steady_clock::time_point start_; + }; + + // Presented frames over the last half second. + void update_fps_counter(); + VkInstance instance_ = VK_NULL_HANDLE; + VkSurfaceKHR surface_ = VK_NULL_HANDLE; + VkPhysicalDevice physical_ = VK_NULL_HANDLE; + VkDevice device_ = VK_NULL_HANDLE; + VkQueue queue_ = VK_NULL_HANDLE; + std::uint32_t queue_family_ = 0; + std::string device_name_; + VkSwapchainKHR swapchain_ = VK_NULL_HANDLE; + VkFormat swapchain_format_ = VK_FORMAT_UNDEFINED; + VkExtent2D extent_{}; + std::vector image_views_; + std::vector swapchain_images_; + void note_input(const SDL_Event& event); + std::chrono::steady_clock::time_point last_input_ = std::chrono::steady_clock::now(); + int fingers_down_ = 0, buttons_down_ = 0; + + // --screenshot-out: read back the swapchain image of one frame into a BMP file. + std::string screenshot_path_; + std::uint64_t screenshot_frame_ = 0; + VkImage depth_image_ = VK_NULL_HANDLE; + VkDeviceMemory depth_memory_ = VK_NULL_HANDLE; + VkImageView depth_view_ = VK_NULL_HANDLE; + VkSampleCountFlagBits samples_ = VK_SAMPLE_COUNT_1_BIT; + VkImage msaa_image_ = VK_NULL_HANDLE; + VkDeviceMemory msaa_memory_ = VK_NULL_HANDLE; + VkImageView msaa_view_ = VK_NULL_HANDLE; + VkRenderPass pass_ = VK_NULL_HANDLE; + std::vector framebuffers_; + VkSampler sampler_ = VK_NULL_HANDLE; + VkSampler clamp_sampler_ = VK_NULL_HANDLE; + VkSampler ui_sampler_ = VK_NULL_HANDLE; + std::unordered_map state_samplers_; + VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE; + VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE; + VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE; + // Everything one frame writes while the GPU may still read the previous frame's copy: with + // kFramesInFlight slots the CPU prepares frame N while the GPU draws frame N-1. A slot is reused + // only after its fence (the submit of frame N - frames_in_flight_) has signalled. + struct FrameSlot { + VkCommandBuffer command = VK_NULL_HANDLE; + VkFence fence = VK_NULL_HANDLE; + VkSemaphore acquire = VK_NULL_HANDLE; + std::uint32_t query_base = 0; + bool has_pending_query = false; + bool recording = false; + VkDescriptorSet bone_descriptor_set = VK_NULL_HANDLE; + VkBuffer bone_buffer = VK_NULL_HANDLE; + VkDeviceMemory bone_memory = VK_NULL_HANDLE; + float* bone_mapped = nullptr; + std::size_t bone_capacity = 0; + VkBuffer state_buffer = VK_NULL_HANDLE; + VkDeviceMemory state_memory = VK_NULL_HANDLE; + std::uint8_t* state_mapped = nullptr; + std::size_t state_capacity = 0; + VkBuffer ui_buffer = VK_NULL_HANDLE; + VkDeviceMemory ui_memory = VK_NULL_HANDLE; + std::uint8_t* ui_mapped = nullptr; + std::size_t ui_vertex_capacity = 0, ui_index_capacity = 0; + VkDeviceSize ui_vertex_bytes = 0; + // Texture uploads recorded into this frame's command buffer copy from here. + VkBuffer staging_buffer = VK_NULL_HANDLE; + VkDeviceMemory staging_memory = VK_NULL_HANDLE; + std::uint8_t* staging_mapped = nullptr; + VkDeviceSize staging_capacity = 0, staging_used = 0, staging_wanted = 0; + // Uploads larger than the ring get their own staging buffer, freed when the slot comes back. + std::vector> retired_buffers; + }; + static constexpr std::uint32_t kMaxFramesInFlight = 2; + std::array frames_{}; + FrameSlot* f_ = &frames_[0]; + std::uint32_t frames_in_flight_ = kMaxFramesInFlight; + std::uint32_t frame_slot_ = 0; + // Signalled by a frame's submit, waited by the present of that swapchain image. + std::vector rendered_; + VkDeviceSize state_stride_ = 0; + VkDeviceSize staging_alignment_ = 16; + GpuTexture fallback_texture_{}; + std::unordered_map textures_; + std::unordered_map paired_descriptors_; + std::unordered_map render_targets_; + VkRenderPass offscreen_pass_ = VK_NULL_HANDLE; + VkFormat offscreen_format_ = VK_FORMAT_R5G6B5_UNORM_PACK16; + std::size_t last_offscreen_draw_count_ = 0; + VkShaderModule vertex_module_ = VK_NULL_HANDLE; + VkShaderModule fragment_module_ = VK_NULL_HANDLE; + VkPipelineLayout layout_ = VK_NULL_HANDLE; + std::unordered_map pipelines_; + std::unordered_map geometries_; + std::uint64_t frame_number_ = 0; + std::uint64_t timed_frames_ = 0; + std::size_t upload_count_ = 0, uploaded_bytes_ = 0; + std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0; + std::string last_texture_upload_name_; + VkCommandPool pool_ = VK_NULL_HANDLE; + VkQueryPool query_pool_ = VK_NULL_HANDLE; + bool os_cursor_hidden_ = false; + bool hardware_cursor_enabled_ = false; + std::array game_cursors_{}; + SDL_Cursor* fallback_cursor_ = nullptr; + int current_cursor_shape_ = -1; + int last_mx_ = 0, last_my_ = 0; + std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0; + std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0; + std::size_t last_vertex_count_ = 0, last_index_count_ = 0; + Timings timings_{}; + double last_sync_ms_ = 0; +}; + +// One summary line of frame and renderer counters on stdout. +void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms = 0.0, + std::vector frame_ms = {}); + +} // namespace mt_host