diff --git a/.agents/skills/metin2-40250-parity-audit/references/project-map.md b/.agents/skills/metin2-40250-parity-audit/references/project-map.md index 0acd48f6..5420a837 100644 --- a/.agents/skills/metin2-40250-parity-audit/references/project-map.md +++ b/.agents/skills/metin2-40250-parity-audit/references/project-map.md @@ -33,7 +33,7 @@ - `src/port//.{h,cpp}` — mirror of each 40250 `logic`/`python` unit (see SKILL.md "Code layout"). - `src/platform/` — implementations of the mirrored 40250 platform classes (D3D8 recorder, Granny, Miles, SpeedTree stand-in, window/input, ScriptLib boot). -- `src/host/` — SDL3 + Vulkan host (`main.cpp` entry, `live_client.cpp` loop, `vulkan_window.cpp` renderer); `src/gr2/` — Granny reader; `src/codepage/` — code pages. +- `src/host/` — SDL3 + Vulkan host (`main.cpp` entry, `live_client.cpp` loop, `vulkan_window.h` renderer split across `vulkan_window.cpp`/`vk_*.cpp`/`sdl_window.cpp`/`ui_batches.cpp`); `src/gr2/` — Granny reader; `src/codepage/` — code pages. - `third_party/` — vendored cpython 2.7.18, cryptopp, freetype, minilzo. - `tests/port/` — port tests and the fake login server; `tests/gr2/` — Granny reader tests. - `android/` — SDLActivity shell, `android/build.sh`, `android/push-client.sh`. diff --git a/audit/port-map/EterLib/GrpImageTexture.cpp.json b/audit/port-map/EterLib/GrpImageTexture.cpp.json index 2633870f..c19151f4 100644 --- a/audit/port-map/EterLib/GrpImageTexture.cpp.json +++ b/audit/port-map/EterLib/GrpImageTexture.cpp.json @@ -29,8 +29,8 @@ "status": "ADAPTED", "impl": [ "src/host/dxt.cpp:decode", - "src/host/vulkan_window.cpp:get_gpu_texture", - "src/host/vulkan_window.cpp:upload_texture" + "src/host/vk_textures.cpp:get_gpu_texture", + "src/host/vk_textures.cpp:upload_texture" ], "test": "tests/port/dxt_decode_test.cpp", "note": "The platform CreateDDSTexture member is a stub; its level logic lives in the texture decode/upload the renderer does. mipmapCount = max(1, min(dwMipMapCount, MAX_MIPLEVELS=12)), level i read at sum(dwLinearSize>>2k) with CDXTCImage::Copy's dwLinearSize>>2i bytes (the rest zero, uninitialised in 40250), decoded to RGBA8 (no GPU DXT) and uploaded as file levels, not regenerated. Divergence: without DDSD_MIPMAPCOUNT, or with DDSD_PITCH, 40250 creates the extra levels uninitialised; here that is one level. IsLowTextureMemory's uTexBias level skip is not applied (the port's ms_isLowTextureMemory is false). The far-terrain look NEEDS_LIVE." @@ -39,7 +39,7 @@ "status": "ADAPTED", "impl": [ "src/platform/EterLib/GrpImageTexture.cpp:CGraphicImageTexture::CreateFromMemoryFile", - "src/host/vulkan_window.cpp:get_gpu_texture" + "src/host/vk_textures.cpp:get_gpu_texture" ], "test": "tests/port/dxt_decode_test.cpp", "note": "The platform body records only the size and file name; the renderer decodes the bytes. 40250 dispatch is kept: a DXT DDS takes the CreateDDSTexture levels, anything else takes D3DXCreateTextureFromFileInMemoryEx(MipLevels=D3DX_DEFAULT), i.e. a full generated chain (vkCmdBlitImage linear, instead of D3DX's box filter). The invented is_ui_tex rule (no mips for icon/, ui/, locale/ and textures <=128px) was removed. Memory textures (mem:, 40250 CreateTexture levels=1) get one level. GRAPHICS_CAPS_HALF_SIZE_IMAGE/low-memory A4R4G4B4 re-encode is not applied." diff --git a/docs/PORT-PLAN.md b/docs/PORT-PLAN.md index 01c36f33..b0d5eaf8 100644 --- a/docs/PORT-PLAN.md +++ b/docs/PORT-PLAN.md @@ -64,7 +64,11 @@ src/platform/ScriptLib/PythonBoot.cpp # 解释器启动、RunMainScript third_party/cpython-2.7.18/ # 静态库 src/host/main.cpp # SDL3 宿主入口:命令行、合成基准 src/host/live_client.cpp # run_live_client:40250 客户端主循环 -src/host/vulkan_window.cpp # SDL 窗口/事件 + Vulkan 渲染器(回放 Render3DDraw / UIRenderCommand) +src/host/vulkan_window.{h,cpp} # VulkanWindow:构造/析构 + 帧计时统计(回放 Render3DDraw / UIRenderCommand) +src/host/sdl_window.cpp # SDL 窗口/事件/光标/屏幕键盘/截图 BMP +src/host/vk_{device,pipelines,resources,textures}.cpp # 交换链 / 渲染通道+管线 / 缓冲+RT+几何 / 纹理+采样器+描述符 +src/host/vk_frame.cpp # render():prepare_draws → prepare_ui → write_draw_states → record_passes → readback +src/host/ui_batches.cpp # UIRenderCommand → UiBatch 合批 src/host/frame_policy.cpp # 帧率策略(FrameRatePolicy) src/host/native_audio.cpp # MilesLib AudioCommand → SDL 音频流 src/host/audio_decode.cpp # .wav(手写 PCM 解析)/.mp3(dr_mp3)→ 44.1 kHz 立体声 float,全平台同一路径 diff --git a/src/host/CMakeLists.txt b/src/host/CMakeLists.txt index 3eb58d68..6d8a91bc 100644 --- a/src/host/CMakeLists.txt +++ b/src/host/CMakeLists.txt @@ -23,6 +23,13 @@ set(MT_NATIVE_RENDER_SOURCES render_state.cpp texture_decode.cpp vulkan_window.cpp + sdl_window.cpp + ui_batches.cpp + vk_device.cpp + vk_frame.cpp + vk_pipelines.cpp + vk_resources.cpp + vk_textures.cpp stb_image_impl.cpp dxt.cpp ) diff --git a/src/host/sdl_window.cpp b/src/host/sdl_window.cpp new file mode 100644 index 00000000..468146ea --- /dev/null +++ b/src/host/sdl_window.cpp @@ -0,0 +1,382 @@ +// The SDL side of VulkanWindow: events and input forwarding, cursor, screen keyboard, logical size, screenshots on disk. +#include "vulkan_window.h" + +#include "input_keymap.h" +#include "texture_decode.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "platform/ScriptLib/PythonBoot.h" +#endif + +#include +#include +#include +#include +#include +#include +#include + +namespace mt_host { + +namespace { + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +SDL_Cursor* load_game_cursor(int shape) { + static constexpr std::array images = { + "cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub", + "cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub", + "cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub", + "cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"}; + if (shape < 0 || shape >= static_cast(images.size())) return nullptr; + const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape]; + std::vector sub_bytes; + if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr; + std::unordered_map tokens; + std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end())); + std::string line; + while (std::getline(input, line)) { + std::istringstream fields(line); + std::string key, value; + if (!(fields >> key >> value)) continue; + if (!value.empty() && value.front() == '"') { + value.erase(0, 1); + if (!value.empty() && value.back() == '"') value.pop_back(); + } + std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); }); + tokens[key] = value; + } + if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr; + const std::string image_path = tokens["version"] == "2.0" + ? std::string("d:/ymir work/ui/cursor/") + tokens["image"] + : std::string("d:/ymir work/ui/") + tokens["image"]; + std::vector image_bytes; + if (!read_live_pack_texture(image_path, image_bytes)) return nullptr; + const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size()); + if (!image.ok()) return nullptr; + const int left = std::atoi(tokens["left"].c_str()); + const int top = std::atoi(tokens["top"].c_str()); + const int right = std::atoi(tokens["right"].c_str()); + const int bottom = std::atoi(tokens["bottom"].c_str()); + if (left < 0 || top < 0 || right <= left || bottom <= top || + right > image.w || bottom > image.h) return nullptr; + const int width = right - left, height = bottom - top; + std::vector pixels(std::size_t(width) * height * 4); + for (int y = 0; y < height; ++y) + std::memcpy(pixels.data() + std::size_t(y) * width * 4, + image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4, + std::size_t(width) * 4); + SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32, + pixels.data(), width * 4); + if (!surface) return nullptr; + const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0; + const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0; + SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y); + SDL_DestroySurface(surface); + return cursor; +} +#endif + +} // namespace + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + +void VulkanWindow::enable_game_hardware_cursor() { + hardware_cursor_enabled_ = true; + for (int shape = 0; shape < static_cast(game_cursors_.size()); ++shape) { + game_cursors_[shape] = load_game_cursor(shape); + } + fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT); + SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_); + SDL_ShowCursor(); + os_cursor_hidden_ = false; +} + +void VulkanWindow::sync_game_cursor() { + if (!hardware_cursor_enabled_) return; + const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() && + !PythonBoot::IsSoftwareCursorVisible(); + if (visible == os_cursor_hidden_) { + if (visible) SDL_ShowCursor(); + else SDL_HideCursor(); + os_cursor_hidden_ = !visible; + } + if (!visible) return; + const int shape = PythonBoot::CursorShape(); + if (shape == current_cursor_shape_) return; + SDL_Cursor* cursor = shape >= 0 && shape < static_cast(game_cursors_.size()) + ? game_cursors_[shape] : nullptr; + if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_); + current_cursor_shape_ = shape; +} +#endif + +void VulkanWindow::write_bmp(const std::string& path, const std::uint8_t* pixels) const { + const std::uint32_t w = extent_.width, h = extent_.height; + const bool rgba = swapchain_format_ == VK_FORMAT_R8G8B8A8_UNORM || swapchain_format_ == VK_FORMAT_R8G8B8A8_SRGB; + std::vector file(54 + std::size_t(w) * h * 4); + auto put32 = [&](std::size_t at, std::uint32_t value) { std::memcpy(&file[at], &value, 4); }; + file[0] = 'B'; file[1] = 'M'; + put32(2, std::uint32_t(file.size())); put32(10, 54); put32(14, 40); + put32(18, w); put32(22, h); put32(26, 1u | (32u << 16)); put32(34, w * h * 4); + for (std::uint32_t y = 0; y < h; ++y) + for (std::uint32_t x = 0; x < w; ++x) { + const std::uint8_t* src = pixels + (std::size_t(y) * w + x) * 4; + std::uint8_t* dst = &file[54 + (std::size_t(h - 1 - y) * w + x) * 4]; + dst[0] = rgba ? src[2] : src[0]; dst[1] = src[1]; dst[2] = rgba ? src[0] : src[2]; dst[3] = 255; + } + std::ofstream out(path, std::ios::binary); + out.write(reinterpret_cast(file.data()), std::streamsize(file.size())); +} + +double VulkanWindow::input_idle_seconds() const { + if (fingers_down_ > 0 || buttons_down_ > 0) return 0.0; + return std::chrono::duration(std::chrono::steady_clock::now() - last_input_).count(); +} + +void VulkanWindow::request_screenshot(std::string path, std::uint64_t frame) { + screenshot_path_ = std::move(path); + screenshot_frame_ = frame; +} + +std::uint32_t VulkanWindow::logical_width() const { + int w = 0, h = 0; + if (window_) SDL_GetWindowSize(window_, &w, &h); +#ifdef __ANDROID__ + // Keep the 40250 UI at the same height as the macOS 1280x800 window, + // while retaining the device aspect ratio on wider phone displays. + if (w > 0 && h > 0) return std::max(1, int(std::lround(double(w) * 800.0 / h))); +#endif + return w > 0 ? static_cast(w) : extent_.width; +} + +std::uint32_t VulkanWindow::logical_height() const { + int w = 0, h = 0; + if (window_) SDL_GetWindowSize(window_, &w, &h); +#ifdef __ANDROID__ + if (w > 0 && h > 0) return 800; +#endif + return h > 0 ? static_cast(h) : extent_.height; +} + +int VulkanWindow::ui_safe_inset() const { + int pw = 0, ph = 0; + if (safe_inset_px_ <= 0 || !window_ || !SDL_GetWindowSizeInPixels(window_, &pw, &ph) || ph <= 0) return 0; + return int(std::lround(double(safe_inset_px_) * logical_height() / ph)); +} + +void VulkanWindow::sync_screen_keyboard() { + const bool want = touch_controller.wants_screen_keyboard(); + const unsigned serial = touch_controller.screen_keyboard_serial(); + if (want && (!SDL_TextInputActive(window_) || serial != screen_keyboard_serial_)) { + SDL_PropertiesID props = SDL_CreateProperties(); + SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_TYPE_NUMBER, SDL_TEXTINPUT_TYPE_TEXT); + SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_CAPITALIZATION_NUMBER, SDL_CAPITALIZE_NONE); + SDL_SetBooleanProperty(props, SDL_PROP_TEXTINPUT_AUTOCORRECT_BOOLEAN, false); + SDL_StartTextInputWithProperties(window_, props); + SDL_DestroyProperties(props); + } else if (!want && SDL_TextInputActive(window_)) { + SDL_StopTextInput(window_); + } + screen_keyboard_serial_ = serial; +} + +bool VulkanWindow::poll(bool forward_to_live_client) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + auto map_mouse = [&](float wx, float wy, int& ox, int& oy) { + int win_w = 0, win_h = 0; + SDL_GetWindowSize(window_, &win_w, &win_h); + unsigned ui_w = extent_.width ? extent_.width : 960; + unsigned ui_h = extent_.height ? extent_.height : 640; + UIRenderGetSize(&ui_w, &ui_h); + if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) { + ox = static_cast(std::lround(wx * float(ui_w) / float(win_w))); + oy = static_cast(std::lround(wy * float(ui_h) / float(win_h))); + } else { + ox = static_cast(std::lround(wx)); + oy = static_cast(std::lround(wy)); + } + }; + std::optional> pending_mouse_move; + auto flush_mouse_move = [&] { + if (!pending_mouse_move) return; + const auto [mx, my] = *pending_mouse_move; + pending_mouse_move.reset(); + last_mx_ = mx; + last_my_ = my; + if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return; + PythonBoot::UIMouseMove(mx, my); + }; +#endif + SDL_Event event; + while (SDL_PollEvent(&event)) { + note_input(event); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + // Touches arrive as SDL_EVENT_FINGER_*; the touch-synthesized mouse copy + // would register as a second finger (101) and turn every drag into a pinch. + if (touch_controller.is_enabled() && + ((event.type == SDL_EVENT_MOUSE_MOTION && event.motion.which == SDL_TOUCH_MOUSEID) || + ((event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) && + event.button.which == SDL_TOUCH_MOUSEID))) + continue; + if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) { + int mx = 0, my = 0; + map_mouse(event.motion.x, event.motion.y, mx, my); + pending_mouse_move = std::pair{mx, my}; + continue; + } + // Preserve event order at button/key/focus boundaries while collapsing + // high-frequency motion into the most recent position. + flush_mouse_move(); +#endif + if (event.type == SDL_EVENT_QUIT) return false; + if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) { + int ew = 0, eh = 0; + SDL_GetWindowSize(window_, &ew, &eh); + touch_controller.update_screen_size(int(logical_width()), int(logical_height()), ui_safe_inset()); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && ew > 0 && eh > 0) { + PythonBoot::SetUISafeInset(ui_safe_inset()); + PythonBoot::SetUISize(int(logical_width()), int(logical_height())); + } +#endif + recreate_swapchain(); + continue; + } + if (event.type == SDL_EVENT_FINGER_DOWN) { + touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } else if (event.type == SDL_EVENT_FINGER_MOTION) { + touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) { + touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client) { + if (!hardware_cursor_enabled_ && + (event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) { + std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n"); + SDL_HideCursor(); + os_cursor_hidden_ = true; + } else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) { + SDL_CaptureMouse(false); + PythonBoot::UIMouseButton(2, false, last_mx_, last_my_); + PythonBoot::UIMouseButton(3, false, last_mx_, last_my_); + PythonBoot::UIMouseButton(1, false, last_mx_, last_my_); + } + if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) { + const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN; + int btn = 0; + if (event.button.button == SDL_BUTTON_LEFT) btn = 1; + else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2; + else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3; + if (btn) { + int mx = 0, my = 0; + map_mouse(event.button.x, event.button.y, mx, my); + last_mx_ = mx; + last_my_ = my; + if (touch_controller.is_enabled()) { + if (touch_controller.on_mouse_button(btn, pressed, mx, my)) { + continue; + } + } + SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0); + PythonBoot::UIMouseButton(btn, pressed, mx, my); + } + } else if (event.type == SDL_EVENT_MOUSE_WHEEL) { + PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f)); + } else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) { + const bool pressed = event.type == SDL_EVENT_KEY_DOWN; + // Android back = ESC: closes the top window, or opens the system menu (game.py). + if (event.key.scancode == SDL_SCANCODE_AC_BACK) event.key.scancode = SDL_SCANCODE_ESCAPE; + if (pressed) { + if (const int vk = sdl_scancode_to_vk(event.key.scancode)) + PythonBoot::UIIMEKeyDown(vk); + if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8); + else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9); + else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER) + PythonBoot::UIChar(13); + else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27); + } + if (!event.key.repeat) { + if (const int dik = sdl_scancode_to_dik(event.key.scancode)) + PythonBoot::UIKey(dik, pressed); + } + } else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) { + const auto* s = reinterpret_cast(event.text.text); + while (*s) { + unsigned cp = 0; + if (*s < 0x80u) { + cp = *s++; + } else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) { + cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu); + s += 2; + } else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) { + cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu); + s += 3; + } else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) { + cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu); + s += 4; + } else { + ++s; + continue; + } + if (cp >= 32u && cp != 127u) + PythonBoot::UIChar(cp); + } + } + } +#else + (void)forward_to_live_client; +#endif + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + flush_mouse_move(); +#endif + touch_controller.update(); + if (touch_controller.is_enabled()) + sync_screen_keyboard(); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) { + SDL_HideCursor(); + os_cursor_hidden_ = true; + } +#endif + return true; +} + +void VulkanWindow::update_fps_counter() { + const auto now = std::chrono::steady_clock::now(); + if (fps_frames_++ == 0) { + fps_window_start_ = now; + return; + } + const double elapsed = std::chrono::duration(now - fps_window_start_).count(); + if (elapsed >= 0.5) { + fps_ = int(std::lround((fps_frames_ - 1) / elapsed)); + fps_frames_ = 1; + fps_window_start_ = now; + } +} + +void VulkanWindow::note_input(const SDL_Event& event) { + switch (event.type) { + case SDL_EVENT_FINGER_DOWN: ++fingers_down_; break; + case SDL_EVENT_FINGER_UP: + case SDL_EVENT_FINGER_CANCELED: fingers_down_ = std::max(0, fingers_down_ - 1); break; + case SDL_EVENT_MOUSE_BUTTON_DOWN: ++buttons_down_; break; + case SDL_EVENT_MOUSE_BUTTON_UP: buttons_down_ = std::max(0, buttons_down_ - 1); break; + case SDL_EVENT_WINDOW_FOCUS_LOST: case SDL_EVENT_WILL_ENTER_BACKGROUND: + // An up event lost with the focus must not keep the client out of the idle rate for good. + fingers_down_ = buttons_down_ = 0; + break; + case SDL_EVENT_FINGER_MOTION: case SDL_EVENT_MOUSE_MOTION: case SDL_EVENT_MOUSE_WHEEL: + case SDL_EVENT_KEY_DOWN: case SDL_EVENT_KEY_UP: case SDL_EVENT_TEXT_INPUT: + case SDL_EVENT_WINDOW_FOCUS_GAINED: case SDL_EVENT_DID_ENTER_FOREGROUND: break; + default: return; + } + last_input_ = std::chrono::steady_clock::now(); +} + +} // namespace mt_host diff --git a/src/host/ui_batches.cpp b/src/host/ui_batches.cpp new file mode 100644 index 00000000..78633de2 --- /dev/null +++ b/src/host/ui_batches.cpp @@ -0,0 +1,194 @@ +// VulkanWindow::build_ui_batches: UIRenderCommand quads into the frame UI buffer, batched by texture and blend. +#include "vulkan_window.h" + +#include +#include +#include + +namespace mt_host { + +void VulkanWindow::build_ui_batches( + std::uint32_t ui_width, + std::uint32_t ui_height, + const std::vector& commands, + const std::unordered_map>& capture_textures, + std::vector& batches, + std::size_t& out_quad_count) { + ensure_ui_capacity(commands.size()); + auto* vertices = reinterpret_cast(f_->ui_mapped); + auto* indices = reinterpret_cast(f_->ui_mapped + f_->ui_vertex_bytes); + std::uint32_t v_count = 0; + std::uint32_t i_count = 0; + const float inv_w = 2.0f / float(ui_width); + const float inv_h = 2.0f / float(ui_height); + + for (const auto& cmd : commands) { + if (v_count + 4 > f_->ui_vertex_capacity || i_count + 6 > f_->ui_index_capacity) + throw std::runtime_error("UI command buffer capacity exceeded"); + if (cmd.kind == UIRenderCommand::Text) continue; + + float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2; + float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev; + const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1; + + float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; + std::array top_color = unpack_argb(cmd.argb); + std::array bot_color = + cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color; + if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue; + + if (cmd.kind == UIRenderCommand::Line) { + const float dx = x2 - x1, dy = y2 - y1; + const float len = std::sqrt(dx * dx + dy * dy); + if (len < 0.1f) continue; + const float nx = -dy / len * 0.5f; + const float ny = dx / len * 0.5f; + qx[0] = x1 + nx; qy[0] = y1 + ny; + qx[1] = x2 + nx; qy[1] = y2 + ny; + qx[2] = x1 - nx; qy[2] = y1 - ny; + qx[3] = x2 - nx; qy[3] = y2 - ny; + for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; } + } else if (cmd.kind == UIRenderCommand::Image && cmd.quad) { + const bool axis_aligned = + std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f && + std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f && + std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f && + std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f; + for (int k = 0; k < 4; ++k) { + mu[k] = cmd.mu[k]; + mv[k] = cmd.mv[k]; + } + if (has_clip && axis_aligned) { + const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3]; + const float cx1 = std::max(ox1, cmd.clip_x1); + const float cy1 = std::max(oy1, cmd.clip_y1); + const float cx2 = std::min(ox2, cmd.clip_x2); + const float cy2 = std::min(oy2, cmd.clip_y2); + if (cx2 <= cx1 || cy2 <= cy1) continue; + float msu = cmd.mu[0], meu = cmd.mu[1]; + float msv = cmd.mv[0], mev = cmd.mv[2]; + if (ox2 > ox1) { + const float t0 = (cx1 - ox1) / (ox2 - ox1); + const float t1 = (cx2 - ox1) / (ox2 - ox1); + const float du = eu - su; + su = cmd.su + du * t0; + eu = cmd.su + du * t1; + const float mdu = meu - msu; + msu = cmd.mu[0] + mdu * t0; + meu = cmd.mu[0] + mdu * t1; + } + if (oy2 > oy1) { + const float t0 = (cy1 - oy1) / (oy2 - oy1); + const float t1 = (cy2 - oy1) / (oy2 - oy1); + const float dv = ev - sv; + sv = cmd.sv + dv * t0; + ev = cmd.sv + dv * t1; + const float mdv = mev - msv; + msv = cmd.mv[0] + mdv * t0; + mev = cmd.mv[0] + mdv * t1; + } + qx[0] = cx1; qy[0] = cy1; + qx[1] = cx2; qy[1] = cy1; + qx[2] = cx1; qy[2] = cy2; + qx[3] = cx2; qy[3] = cy2; + mu[0] = msu; mv[0] = msv; + mu[1] = meu; mv[1] = msv; + mu[2] = msu; mv[2] = mev; + mu[3] = meu; mv[3] = mev; + } else { + if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) + continue; + for (int k = 0; k < 4; ++k) { + qx[k] = cmd.qx[k]; + qy[k] = cmd.qy[k]; + } + } + u[0] = su; v[0] = sv; + u[1] = eu; v[1] = sv; + u[2] = su; v[2] = ev; + u[3] = eu; v[3] = ev; + } else if (cmd.kind == UIRenderCommand::Bar && cmd.quad) { + // An untextured polygon piece (CPythonGraphic::RenderCoolTimeBox's fan triangles). + if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) + continue; + for (int k = 0; k < 4; ++k) { + qx[k] = cmd.qx[k]; + qy[k] = cmd.qy[k]; + u[k] = 0.0f; + v[k] = 0.0f; + } + } else { + if (has_clip) { + x1 = std::max(x1, cmd.clip_x1); + y1 = std::max(y1, cmd.clip_y1); + x2 = std::min(x2, cmd.clip_x2); + y2 = std::min(y2, cmd.clip_y2); + } + if (x2 <= x1 || y2 <= y1) continue; + qx[0] = x1; qy[0] = y1; + qx[1] = x2; qy[1] = y1; + qx[2] = x1; qy[2] = y2; + qx[3] = x2; qy[3] = y2; + u[0] = su; v[0] = sv; + u[1] = eu; v[1] = sv; + u[2] = su; v[2] = ev; + u[3] = eu; v[3] = ev; + } + + const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string(); + const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string(); + const bool is_masked = !mask_name.empty(); + // CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use + // INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR. + const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image + ? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1)) + : 1; + const VkPipeline pipeline = get_pipeline(0, 0, blend_mode); + const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures); + + for (int k = 0; k < 4; ++k) { + Vertex& vert = vertices[v_count + k]; + vert.position[0] = qx[k] * inv_w - 1.0f; + vert.position[1] = 1.0f - qy[k] * inv_h; + vert.position[2] = 0.0f; + vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f; + vert.uv[0] = u[k]; vert.uv[1] = v[k]; + const auto& col = (k < 2) ? top_color : bot_color; + vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3]; + vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0; + vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f; + vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k]; + vert.rhw = 1.0f; + } + indices[i_count + 0] = v_count + 0; + indices[i_count + 1] = v_count + 1; + indices[i_count + 2] = v_count + 2; + indices[i_count + 3] = v_count + 2; + indices[i_count + 4] = v_count + 1; + indices[i_count + 3 + 2] = v_count + 3; + + const float mask_flag = is_masked ? 1.0f : 0.0f; + if (!batches.empty() && + batches.back().behind_3d == cmd.behind_3d && + batches.back().pipeline == pipeline && + batches.back().descriptor == descriptor && + batches.back().constants.params[2] == mask_flag) { + batches.back().index_count += 6; + } else { + UiBatch batch{}; + batch.first_index = i_count; + batch.index_count = 6; + batch.pipeline = pipeline; + batch.descriptor = descriptor; + batch.behind_3d = cmd.behind_3d; + batch.constants.mvp = kIdentityMatrix; + batch.constants.params = {0.0f, 0.0f, mask_flag, 0.0f}; + batches.push_back(batch); + } + v_count += 4; + i_count += 6; + ++out_quad_count; + } +} + +} // namespace mt_host diff --git a/src/host/vk_device.cpp b/src/host/vk_device.cpp new file mode 100644 index 00000000..26ab8c0b --- /dev/null +++ b/src/host/vk_device.cpp @@ -0,0 +1,251 @@ +// VulkanWindow: physical device selection, swapchain, MSAA target and framebuffers. +#include "vulkan_window.h" + +#include "host_util.h" + +#include +#include +#include +#include + +namespace mt_host { + +std::uint32_t VulkanWindow::find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const { + VkPhysicalDeviceMemoryProperties properties{}; + vkGetPhysicalDeviceMemoryProperties(physical_, &properties); + for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i) + if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags) + return i; + throw std::runtime_error("no matching Vulkan memory type"); +} + +void VulkanWindow::select_device() { + std::uint32_t count = 0; + check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count"); + if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK"); + std::vector candidates(count); + check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices"); + for (auto candidate : candidates) { + std::uint32_t family_count = 0; + vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr); + std::vector families(family_count); + vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data()); + for (std::uint32_t i = 0; i < family_count; ++i) { + VkBool32 present = VK_FALSE; + check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support"); + if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) { + physical_ = candidate; queue_family_ = i; break; + } + } + if (physical_) break; + } + if (!physical_) throw std::runtime_error("no graphics/present queue family"); + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + device_name_ = properties.deviceName; + timestamp_period_ns_ = properties.limits.timestampPeriod; + VkPhysicalDeviceFeatures supported_features{}; + vkGetPhysicalDeviceFeatures(physical_, &supported_features); + VkPhysicalDeviceFeatures enabled_features{}; + if (supported_features.samplerAnisotropy) { + enabled_features.samplerAnisotropy = VK_TRUE; + max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy); + } + // Multisampled 3D edges; MT_MSAA=1 keeps the single-sample path, 2/4/8 request a count. + { + int wanted = 4; + if (const char* env = std::getenv("MT_MSAA")) wanted = std::max(1, std::atoi(env)); + const VkSampleCountFlags supported = + properties.limits.framebufferColorSampleCounts & properties.limits.framebufferDepthSampleCounts; + samples_ = VK_SAMPLE_COUNT_1_BIT; + for (VkSampleCountFlagBits c : {VK_SAMPLE_COUNT_8_BIT, VK_SAMPLE_COUNT_4_BIT, VK_SAMPLE_COUNT_2_BIT}) + if (int(c) <= wanted && (supported & c)) { samples_ = c; break; } + } + std::uint32_t extension_count = 0; + check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count"); + std::vector available(extension_count); + check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions"); + std::vector extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME}; + constexpr const char* portability_subset = "VK_KHR_portability_subset"; + for (const auto& extension : available) + if (std::strcmp(extension.extensionName, portability_subset) == 0) + extensions.push_back(portability_subset); + const float priority = 1.0f; + VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; + queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority; + VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; + device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info; + device_info.enabledExtensionCount = static_cast(extensions.size()); + device_info.ppEnabledExtensionNames = extensions.data(); + device_info.pEnabledFeatures = &enabled_features; + check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice"); + vkGetDeviceQueue(device_, queue_family_, 0, &queue_); +} + +void VulkanWindow::create_swapchain() { + VkSurfaceCapabilitiesKHR capabilities{}; + check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities"); + std::uint32_t count = 0; + check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count"); + if (!count) throw std::runtime_error("surface has no formats"); + std::vector formats(count); + check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats"); + VkSurfaceFormatKHR format = formats.front(); + for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate; + swapchain_format_ = format.format; + int width = 0, height = 0; + SDL_GetWindowSizeInPixels(window_, &width, &height); + if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent; + else { + extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width); + extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height); + } + + present_mode_ = VK_PRESENT_MODE_FIFO_KHR; + if (!vsync_) { + std::uint32_t mode_count = 0; + check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count"); + std::vector modes(mode_count); + if (mode_count) + check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes"); + if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end()) + present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR; + else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end()) + present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR; + } + + VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR}; + info.surface = surface_; + info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX); + info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_; + info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | + (capabilities.supportedUsageFlags & VK_IMAGE_USAGE_TRANSFER_SRC_BIT); + info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; + // Android surfaces can report a quarter-turn transform even when SDL's window is + // already landscape. Rendering in SDL window coordinates and requesting that transform + // again rotates the entire game image on presentation. + info.preTransform = (capabilities.supportedTransforms & VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR) + ? VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR : capabilities.currentTransform; +#ifdef __ANDROID__ + SDL_Log("swapchain: window=%dx%d extent=%ux%u currentTransform=%u preTransform=%u", + width, height, extent_.width, extent_.height, + unsigned(capabilities.currentTransform), unsigned(info.preTransform)); +#endif + info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_; + info.clipped = VK_TRUE; + check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR"); + check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count"); + std::vector images(count); + check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images"); + swapchain_images_ = images; + for (auto image : images) { + VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; + view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; + VkImageView handle = VK_NULL_HANDLE; + check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView"); + image_views_.push_back(handle); + } + VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + depth_info.imageType = VK_IMAGE_TYPE_2D; + depth_info.format = VK_FORMAT_D32_SFLOAT; + depth_info.extent = {extent_.width, extent_.height, 1}; + depth_info.mipLevels = 1; depth_info.arrayLayers = 1; + depth_info.samples = samples_; + depth_info.tiling = VK_IMAGE_TILING_OPTIMAL; + depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT; + depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth"); + VkMemoryRequirements depth_reqs{}; + vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs); + VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + depth_alloc.allocationSize = depth_reqs.size; + depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth"); + check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth"); + VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D; + depth_view.format = VK_FORMAT_D32_SFLOAT; + depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; + depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1; + check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth"); + if (samples_ != VK_SAMPLE_COUNT_1_BIT) create_msaa_color(); +} + +void VulkanWindow::recreate_swapchain() { + int w = 0, h = 0; + SDL_GetWindowSizeInPixels(window_, &w, &h); + if (w <= 0 || h <= 0) return; + vkDeviceWaitIdle(device_); + for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); + pipelines_.clear(); + for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr); + framebuffers_.clear(); + if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; } + if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; } + if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; } + destroy_msaa_color(); + for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); + image_views_.clear(); + if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; } + create_swapchain(); + create_framebuffers(); + get_pipeline(0, 2, 0); +} + +void VulkanWindow::create_msaa_color() { + VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + info.imageType = VK_IMAGE_TYPE_2D; + info.format = swapchain_format_; + info.extent = {extent_.width, extent_.height, 1}; + info.mipLevels = 1; info.arrayLayers = 1; + info.samples = samples_; + info.tiling = VK_IMAGE_TILING_OPTIMAL; + info.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSIENT_ATTACHMENT_BIT; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &info, nullptr, &msaa_image_), "vkCreateImage msaa color"); + VkMemoryRequirements reqs{}; + vkGetImageMemoryRequirements(device_, msaa_image_, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + try { + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_LAZILY_ALLOCATED_BIT); + } catch (const std::runtime_error&) { + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + } + check(vkAllocateMemory(device_, &alloc, nullptr, &msaa_memory_), "vkAllocateMemory msaa color"); + check(vkBindImageMemory(device_, msaa_image_, msaa_memory_, 0), "vkBindImageMemory msaa color"); + VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view.image = msaa_image_; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; + view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; + check(vkCreateImageView(device_, &view, nullptr, &msaa_view_), "vkCreateImageView msaa color"); +} + +void VulkanWindow::destroy_msaa_color() { + if (msaa_view_) { vkDestroyImageView(device_, msaa_view_, nullptr); msaa_view_ = VK_NULL_HANDLE; } + if (msaa_image_) { vkDestroyImage(device_, msaa_image_, nullptr); msaa_image_ = VK_NULL_HANDLE; } + if (msaa_memory_) { vkFreeMemory(device_, msaa_memory_, nullptr); msaa_memory_ = VK_NULL_HANDLE; } +} + +void VulkanWindow::create_framebuffers() { + for (auto view : image_views_) { + // Single-sample: {swapchain, depth}. MSAA: {msaa colour, depth, swapchain resolve}. + VkImageView fb_attachments[3] = {view, depth_view_, VK_NULL_HANDLE}; + std::uint32_t count = 2; + if (samples_ != VK_SAMPLE_COUNT_1_BIT) { + fb_attachments[0] = msaa_view_; + fb_attachments[2] = view; + count = 3; + } + VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + framebuffer.renderPass = pass_; framebuffer.attachmentCount = count; framebuffer.pAttachments = fb_attachments; + framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; + VkFramebuffer handle = VK_NULL_HANDLE; + check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer"); + framebuffers_.push_back(handle); + } +} + +} // namespace mt_host diff --git a/src/host/vk_frame.cpp b/src/host/vk_frame.cpp new file mode 100644 index 00000000..e3b837e2 --- /dev/null +++ b/src/host/vk_frame.cpp @@ -0,0 +1,550 @@ +// VulkanWindow::render: one frame from the recorded D3D8 draws and UI commands to present. +#include "vulkan_window.h" + +#include "host_util.h" + +#include +#include +#include +#include +#include +#include + +namespace mt_host { + +VkViewport VulkanWindow::full_viewport() const { + return {0, 0, float(extent_.width), float(extent_.height), 0, 1}; +} + +VkViewport VulkanWindow::draw_viewport(const Render3DDraw& draw) const { + if (draw.viewport[2] <= 0 || draw.viewport[3] <= 0) return full_viewport(); + const float sx = float(extent_.width) / float(logical_width()); + const float sy = float(extent_.height) / float(logical_height()); + return {(draw.viewport[0] + draw.ui_offset_x) * sx, draw.viewport[1] * sy, draw.viewport[2] * sx, draw.viewport[3] * sy, + draw.viewport_z[0], draw.viewport_z[1]}; +} + +void VulkanWindow::render( + const std::vector& draws, + const std::unordered_map>& capture_textures, + std::uint32_t ui_width, + std::uint32_t ui_height, + const std::vector& ui_commands) { + if (extent_.width == 0 || extent_.height == 0) return; + const auto start = std::chrono::steady_clock::now(); + f_ = &frames_[frame_slot_ % frames_in_flight_]; + check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences"); + collect_pending_gpu_timestamp(*f_); + reset_staging(*f_); + const auto fenced = std::chrono::steady_clock::now(); + prune_textures(); + std::uint32_t image_index = 0; + const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, f_->acquire, VK_NULL_HANDLE, &image_index); + if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { + recreate_swapchain(); + return; + } + if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR"); + const auto synchronized = std::chrono::steady_clock::now(); + ++frame_number_; + ++timed_frames_; + while (rendered_.size() < swapchain_images_.size()) { + VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + VkSemaphore semaphore = VK_NULL_HANDLE; + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &semaphore), "vkCreateSemaphore rendered"); + rendered_.push_back(semaphore); + } + // Texture uploads of this frame are recorded ahead of the render pass (upload_texture). + check(vkResetCommandBuffer(f_->command, 0), "vkResetCommandBuffer"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(f_->command, &begin), "vkBeginCommandBuffer"); + vkCmdResetQueryPool(f_->command, query_pool_, f_->query_base, 2); + vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, f_->query_base); + f_->recording = true; + + FrameDraws frame; + prepare_draws(draws, capture_textures, frame); + std::vector ui_batches; + std::size_t ui_quad_count = 0; + prepare_ui(ui_width, ui_height, ui_commands, capture_textures, ui_batches, ui_quad_count); + write_draw_states(frame, ui_batches); + + const auto prepared = std::chrono::steady_clock::now(); + f_->recording = false; + check(vkResetFences(device_, 1, &f_->fence), "vkResetFences"); + + record_passes(image_index, frame, ui_batches); + Readback readback = record_readbacks(image_index); + vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, f_->query_base + 1); + check(vkEndCommandBuffer(f_->command), "vkEndCommandBuffer"); + f_->has_pending_query = true; + + const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &f_->acquire; submit.pWaitDstStageMask = &stage; + submit.commandBufferCount = 1; submit.pCommandBuffers = &f_->command; + submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_[image_index]; + check(vkQueueSubmit(queue_, 1, &submit, f_->fence), "vkQueueSubmit"); + ++frame_slot_; + finish_readbacks(readback); + const auto submitted = std::chrono::steady_clock::now(); + VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; + present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_[image_index]; + present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; + const auto presented = vkQueuePresentKHR(queue_, &present); + // An Android surface can continuously report SUBOPTIMAL while the app deliberately + // uses the identity transform. The image is still presentable; rebuilding every frame + // stalls rendering without changing that status. + if (presented == VK_ERROR_OUT_OF_DATE_KHR) { + recreate_swapchain(); + } else if (presented != VK_SUCCESS && presented != VK_SUBOPTIMAL_KHR) { + check(presented, "vkQueuePresentKHR"); + } + const auto presented_at = std::chrono::steady_clock::now(); + const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); + last_sync_ms_ = std::chrono::duration(synchronized - start).count(); + timings_.sync_ms += last_sync_ms_; + timings_.fence_ms += std::chrono::duration(fenced - start).count(); + timings_.prepare_ms += prepare_ms; + if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; + timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); + timings_.present_ms += std::chrono::duration(presented_at - submitted).count(); + last_draw_count_ = frame.prepared_draws.size(); + last_skinned_draw_count_ = frame.skinned_draw_count; + last_ui_batch_count_ = ui_batches.size(); + last_ui_quad_count_ = ui_quad_count; + last_vertex_count_ = frame.vertex_count; + last_index_count_ = frame.index_count; +} + +void VulkanWindow::prepare_draws(const std::vector& draws, const TextureBytes& capture_textures, FrameDraws& frame) { + auto& prepared_draws = frame.prepared_draws; + auto& offscreen_draws = frame.offscreen_draws; + prepared_draws.reserve(draws.size()); + std::unordered_set active_keys; + std::uint32_t bone_cursor = 0; + std::size_t required_bones = 0; + for (const auto& draw : draws) { + if (!draw.bone_matrices.empty()) { + if (draw.bone_matrices.size() % 16) + throw std::runtime_error("invalid bone palette length"); + required_bones += draw.bone_matrices.size() / 16; + } + } + ensure_bone_capacity(required_bones); + + for (const auto& draw : draws) { + RenderTarget* target = nullptr; + if (draw.render_target) { + target = get_render_target( + Render3DRenderTargetName(draw.render_target, draw.target_width, draw.target_height)); + if (!target) continue; + } + if (draw.clear_flags && target) { + PreparedDraw clear{}; + clear.target = target; + clear.clear_flags = draw.clear_flags; + const auto color = unpack_argb(draw.clear_color); + clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; + clear.clear_depth.depthStencil = {draw.clear_z, 0}; + clear.clear_rect = {{0, 0}, {target->width, target->height}}; + offscreen_draws.push_back(clear); + continue; + } + if (draw.clear_flags) { + PreparedDraw clear{}; + clear.clear_flags = draw.clear_flags; + const auto color = unpack_argb(draw.clear_color); + clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; + clear.clear_depth.depthStencil = {draw.clear_z, 0}; + // D3D8 Clear with no rectangles clears the current viewport. + const auto viewport = draw_viewport(draw); + const float x0 = std::clamp(viewport.x, 0.0f, float(extent_.width)); + const float y0 = std::clamp(viewport.y, 0.0f, float(extent_.height)); + const float x1 = std::clamp(viewport.x + viewport.width, 0.0f, float(extent_.width)); + const float y1 = std::clamp(viewport.y + viewport.height, 0.0f, float(extent_.height)); + clear.clear_rect = {{std::int32_t(x0), std::int32_t(y0)}, + {std::uint32_t(x1 - x0), std::uint32_t(y1 - y0)}}; + prepared_draws.push_back(clear); + continue; + } + if (draw.positions.empty() || draw.indices.empty()) continue; + const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw); + const GeometryId id{draw.geometry_key, signature}; + auto [it, inserted] = geometries_.try_emplace(id); + auto& geometry = it->second; + if (inserted) { + upload_geometry(draw, geometry); + } else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) { + throw std::runtime_error("geometry key/revision reused with a different size"); + } + geometry.last_used_frame = frame_number_; + if (draw.geometry_key) active_keys.insert(draw.geometry_key); + + float skin_offset_encoded = 0.0f; + if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && f_->bone_mapped) { + const auto bone_count = static_cast(draw.bone_matrices.size() / 16); + if (std::size_t(bone_cursor) + bone_count <= f_->bone_capacity) { + std::memcpy( + f_->bone_mapped + std::size_t(bone_cursor) * 16, + draw.bone_matrices.data(), + draw.bone_matrices.size() * sizeof(float)); + skin_offset_encoded = float(bone_cursor + 1u); + bone_cursor += bone_count; + ++frame.skinned_draw_count; + } else throw std::runtime_error("bone palette capacity exceeded"); + } + + const bool is_alpha = draw.alpha_blend != 0; + std::uint8_t cull = 0; + // D3D8 cull mode names describe the winding to discard. With the + // shader's Y flip, clockwise is Vulkan's front-facing winding. + if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise + else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise + const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0; + std::uint8_t blend = 0; + if (is_alpha) { + if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 && + draw.dest_blend >= 1 && draw.dest_blend <= 11) { + blend = static_cast((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4)); + } else { + blend = 1; + } + } + const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, + static_cast(draw.z_func), target != nullptr); + // D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples. + const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1; + const auto descriptor = get_texture_descriptor( + draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw); + + auto constants = make_push_constants(draw, skin_offset_encoded); + if (draw.pretransformed) { + const float screen_width = float(logical_width()); + const float screen_height = float(logical_height()); + // Screen pixels (y down) to D3D NDC (y up); the shader's Y flip then maps to Vulkan. + constants.mvp = {2.0f / screen_width, 0, 0, 0, + 0, -2.0f / screen_height, 0, 0, + 0, 0, 1, 0, + -1, 1, 0, 1}; + } + PreparedDraw prepared_draw{}; + prepared_draw.target = target; + prepared_draw.geometry = &geometry; + prepared_draw.pipeline = pipeline; + prepared_draw.descriptor = descriptor; + prepared_draw.constants = constants; + prepared_draw.fixed_state = make_fixed_function_state(draw); + // XYZRHW vertices are already in screen space and bypass the viewport transform. + prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw); + if (draw.pretransformed && draw.ui_offset_x != 0) + prepared_draw.viewport.x += draw.ui_offset_x * float(extent_.width) / float(logical_width()); + if (target) { + // D3DVIEWPORT8 in the target's own pixels. + prepared_draw.viewport = draw.viewport[2] > 0 && draw.viewport[3] > 0 + ? VkViewport{draw.viewport[0], draw.viewport[1], draw.viewport[2], draw.viewport[3], + draw.viewport_z[0], draw.viewport_z[1]} + : VkViewport{0, 0, float(target->width), float(target->height), 0, 1}; + offscreen_draws.push_back(prepared_draw); + } else prepared_draws.push_back(prepared_draw); + frame.vertex_count += geometry.vertex_count; + frame.index_count += geometry.index_count; + } + for (auto it = geometries_.begin(); it != geometries_.end();) { + const auto age = frame_number_ - it->second.last_used_frame; + const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key); + if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) { + release_geometry(it->second); + it = geometries_.erase(it); + } else ++it; + } +} + +void VulkanWindow::prepare_ui(std::uint32_t ui_width, std::uint32_t ui_height, const std::vector& ui_commands, + const TextureBytes& capture_textures, std::vector& ui_batches, std::size_t& ui_quad_count) { + if (f_->ui_mapped) { + const uint32_t target_ui_w = ui_width ? ui_width : 960; + const uint32_t target_ui_h = ui_height ? ui_height : 640; + update_fps_counter(); + if (touch_controller.is_enabled() || show_fps_) { + std::vector combined_ui = ui_commands; + if (touch_controller.is_enabled()) { + touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h), ui_safe_inset()); + touch_controller.append_ui_commands(combined_ui); + } + if (show_fps_ && fps_ >= 0) { + // Top left, inside the safe area and clear of 40250's corner icon. + const float inset = float(std::min(ui_safe_inset(), int(target_ui_w) / 8)); + TouchController::draw_segment_text(combined_ui, "FPS " + std::to_string(fps_), + inset + 40.0f, 8.0f, 12.0f, 0xFF40FF40); + } + if (!combined_ui.empty()) { + build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count); + } + } else if (!ui_commands.empty()) { + build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count); + } + } +} + +void VulkanWindow::write_draw_states(FrameDraws& frame, std::vector& ui_batches) { + ensure_state_capacity(frame.prepared_draws.size() + frame.offscreen_draws.size() + ui_batches.size()); + std::size_t state_index = 0; + for (auto& draw : frame.offscreen_draws) { + if (!draw.geometry) continue; + draw.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + ++state_index; + } + last_offscreen_draw_count_ = frame.offscreen_draws.size(); + for (auto& draw : frame.prepared_draws) { + if (!draw.geometry) continue; + draw.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + ++state_index; + } + const FixedFunctionState ui_state{}; + for (auto& batch : ui_batches) { + batch.state_offset = static_cast(state_index * state_stride_); + std::memcpy(f_->state_mapped + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); + ++state_index; + } +} + +void VulkanWindow::record_passes(std::uint32_t image_index, const FrameDraws& frame, const std::vector& ui_batches) { + VkClearValue clears[3]{}; + // 40250 clears the back buffer to 0xff000000 once at device creation and afterwards + // only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer). + clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}}; + clears[1].depthStencil = {1.0f, 0}; + VkPipeline current_pipeline = VK_NULL_HANDLE; + VkDescriptorSet current_descriptor = VK_NULL_HANDLE; + + VkViewport current_viewport{-1, -1, -1, -1, -1, -1}; + auto set_viewport = [&](const VkViewport& viewport) { + if (std::memcmp(&viewport, ¤t_viewport, sizeof(viewport)) == 0) return; + vkCmdSetViewport(f_->command, 0, 1, &viewport); + current_viewport = viewport; + }; + + // One prepared draw or Clear inside the current render pass. + auto record_prepared = [&](const PreparedDraw& prepared_draw) { + if (!prepared_draw.geometry) { + VkClearAttachment attachments[2]{}; + std::uint32_t attachment_count = 0; + if (prepared_draw.clear_flags & 1u) { // D3DCLEAR_TARGET + attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + attachments[attachment_count].colorAttachment = 0; + attachments[attachment_count].clearValue = prepared_draw.clear_color; + ++attachment_count; + } + if (prepared_draw.clear_flags & 2u) { // D3DCLEAR_ZBUFFER + attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; + attachments[attachment_count].clearValue = prepared_draw.clear_depth; + ++attachment_count; + } + const VkClearRect rect{prepared_draw.clear_rect, 0, 1}; + if (attachment_count && rect.rect.extent.width && rect.rect.extent.height) + vkCmdClearAttachments(f_->command, attachment_count, attachments, 1, &rect); + return; + } + set_viewport(prepared_draw.viewport); + if (prepared_draw.pipeline != current_pipeline) { + vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); + current_pipeline = prepared_draw.pipeline; + } + if (prepared_draw.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); + current_descriptor = prepared_draw.descriptor; + } + vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &f_->bone_descriptor_set, 1, &prepared_draw.state_offset); + const auto& geometry = *prepared_draw.geometry; + const VkDeviceSize vertex_offset = 0; + vkCmdBindVertexBuffers(f_->command, 0, 1, &geometry.buffer, &vertex_offset); + vkCmdBindIndexBuffer(f_->command, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); + vkCmdPushConstants( + f_->command, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(prepared_draw.constants), + &prepared_draw.constants); + vkCmdDrawIndexed(f_->command, geometry.index_count, 1, 0, 0, 0); + }; + + // Render targets first used this frame (drawn or only sampled) start cleared. + for (auto& entry : render_targets_) + if (!entry.second.initialized) initialize_render_target(entry.second); + // Offscreen passes: one per run of draws into the same target. + for (std::size_t i = 0; i < frame.offscreen_draws.size();) { + RenderTarget* target = frame.offscreen_draws[i].target; + VkRenderPassBeginInfo offscreen_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + offscreen_begin.renderPass = offscreen_pass_; + offscreen_begin.framebuffer = target->framebuffer; + offscreen_begin.renderArea.extent = {target->width, target->height}; + vkCmdBeginRenderPass(f_->command, &offscreen_begin, VK_SUBPASS_CONTENTS_INLINE); + for (; i < frame.offscreen_draws.size() && frame.offscreen_draws[i].target == target; ++i) + record_prepared(frame.offscreen_draws[i]); + vkCmdEndRenderPass(f_->command); + } + + VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); + pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears; + vkCmdBeginRenderPass(f_->command, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); + + auto record_ui_pass = [&](bool behind_3d) { + bool ui_bound = false; + set_viewport(full_viewport()); + for (const auto& batch : ui_batches) { + if (batch.behind_3d != behind_3d) continue; + if (!ui_bound) { + const VkDeviceSize v_offset = 0; + vkCmdBindVertexBuffers(f_->command, 0, 1, &f_->ui_buffer, &v_offset); + vkCmdBindIndexBuffer(f_->command, f_->ui_buffer, f_->ui_vertex_bytes, VK_INDEX_TYPE_UINT32); + ui_bound = true; + } + if (batch.pipeline != current_pipeline) { + vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); + current_pipeline = batch.pipeline; + } + if (batch.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); + current_descriptor = batch.descriptor; + } + vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &f_->bone_descriptor_set, 1, &batch.state_offset); + vkCmdPushConstants( + f_->command, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(batch.constants), + &batch.constants); + vkCmdDrawIndexed(f_->command, batch.index_count, 1, batch.first_index, 0, 0); + } + }; + + record_ui_pass(true); + + for (const auto& prepared_draw : frame.prepared_draws) record_prepared(prepared_draw); + + record_ui_pass(false); + + vkCmdEndRenderPass(f_->command); +} + +VulkanWindow::Readback VulkanWindow::record_readbacks(std::uint32_t image_index) { + Readback readback; + readback.screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_; + if (readback.screenshot) { + const VkDeviceSize bytes = VkDeviceSize(extent_.width) * extent_.height * 4; + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = bytes; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; + check(vkCreateBuffer(device_, &info, nullptr, &readback.buffer), "vkCreateBuffer readback"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, readback.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &readback.memory), "vkAllocateMemory readback"); + check(vkBindBufferMemory(device_, readback.buffer, readback.memory, 0), "vkBindBufferMemory readback"); + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.image = swapchain_images_.at(image_index); + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + VkBufferImageCopy region{}; + region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; + region.imageExtent = {extent_.width, extent_.height, 1}; + vkCmdCopyImageToBuffer(f_->command, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback.buffer, 1, ®ion); + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = 0; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + } + // Debug: MT_DUMP_RT=prefix writes each render target of the screenshot frame to _.pgm. + const char* dump_rt = std::getenv("MT_DUMP_RT"); + if (readback.screenshot && dump_rt && *dump_rt) { + readback.dump_rt = dump_rt; + const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; + for (auto& entry : render_targets_) { + RenderTarget& rt = entry.second; + Readback::RtDump dump{VK_NULL_HANDLE, VK_NULL_HANDLE, rt.width, rt.height}; + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = VkDeviceSize(rt.width) * rt.height * texel; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; + check(vkCreateBuffer(device_, &info, nullptr, &dump.buffer), "vkCreateBuffer rt dump"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, dump.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &dump.memory), "vkAllocateMemory rt dump"); + check(vkBindBufferMemory(device_, dump.buffer, dump.memory, 0), "vkBindBufferMemory rt dump"); + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_SHADER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.image = rt.texture.image; + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + VkBufferImageCopy region{}; + region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; + region.imageExtent = {rt.width, rt.height, 1}; + vkCmdCopyImageToBuffer(f_->command, rt.texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dump.buffer, 1, ®ion); + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); + readback.rt_dumps.push_back(dump); + } + } + return readback; +} + +void VulkanWindow::finish_readbacks(Readback& readback) { + if (!readback.screenshot) return; + check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences readback"); + void* mapped = nullptr; + check(vkMapMemory(device_, readback.memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback"); + write_bmp(screenshot_path_, static_cast(mapped)); + vkUnmapMemory(device_, readback.memory); + const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; + for (std::size_t n = 0; n < readback.rt_dumps.size(); ++n) { + const auto& dump = readback.rt_dumps[n]; + void* data = nullptr; + check(vkMapMemory(device_, dump.memory, 0, VK_WHOLE_SIZE, 0, &data), "vkMapMemory rt dump"); + const auto* bytes = static_cast(data); + std::ofstream out(std::string(readback.dump_rt) + "_" + std::to_string(n) + ".pgm", std::ios::binary); + out << "P5\n" << dump.w << " " << dump.h << "\n255\n"; + for (std::size_t i = 0; i < std::size_t(dump.w) * dump.h; ++i) { + std::uint8_t g = texel == 2 ? std::uint8_t(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31) : bytes[i * 4]; + out.put(char(g)); + } + vkUnmapMemory(device_, dump.memory); + vkDestroyBuffer(device_, dump.buffer, nullptr); + vkFreeMemory(device_, dump.memory, nullptr); + } + vkDestroyBuffer(device_, readback.buffer, nullptr); + vkFreeMemory(device_, readback.memory, nullptr); +} + +} // namespace mt_host diff --git a/src/host/vk_pipelines.cpp b/src/host/vk_pipelines.cpp new file mode 100644 index 00000000..4cdc3bea --- /dev/null +++ b/src/host/vk_pipelines.cpp @@ -0,0 +1,277 @@ +// VulkanWindow: render passes, pipeline layout, shader modules and the pipeline cache keyed by D3D8 state. +#include "vulkan_window.h" + +#include "host_util.h" + +#include +#include + +namespace mt_host { + +namespace { + +std::vector read_spirv(const char* name) { + std::size_t size = 0; + void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size); +#ifdef __ANDROID__ + if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size); + if (!bytes) bytes = SDL_LoadFile(name, &size); +#endif + if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name); + if (size == 0 || size % 4) { + SDL_free(bytes); + throw std::runtime_error(std::string("invalid SPIR-V size: ") + name); + } + std::vector words(size / 4); + std::memcpy(words.data(), bytes, size); + SDL_free(bytes); + return words; +} + +} // namespace + +void VulkanWindow::create_render_pass_and_layout() { + const bool msaa = samples_ != VK_SAMPLE_COUNT_1_BIT; + VkAttachmentDescription attachments[3]{}; + attachments[0].format = swapchain_format_; attachments[0].samples = samples_; + attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; + attachments[0].storeOp = msaa ? VK_ATTACHMENT_STORE_OP_DONT_CARE : VK_ATTACHMENT_STORE_OP_STORE; + attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + attachments[0].finalLayout = msaa ? VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL : VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + // MSAA resolve target: the swapchain image. + attachments[2].format = swapchain_format_; attachments[2].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[2].loadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; attachments[2].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[2].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[2].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = samples_; + attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + + VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; + VkAttachmentReference resolve_ref{2, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkSubpassDescription subpass{}; + subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; + subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; + if (msaa) subpass.pResolveAttachments = &resolve_ref; + subpass.pDepthStencilAttachment = &depth_ref; + + VkSubpassDependency dependency{}; + dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0; + dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; + dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; + dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; + + VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; + pass_info.attachmentCount = msaa ? 3 : 2; pass_info.pAttachments = attachments; + pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; + pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency; + check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass"); + create_framebuffers(); + + const auto vert = read_spirv("native.vert.spv"); + const auto frag = read_spirv("native.frag.spv"); + auto module = [&](const std::vector& code) { + VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO}; + info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data(); + VkShaderModule handle = VK_NULL_HANDLE; + check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule"); + return handle; + }; + vertex_module_ = module(vert); + fragment_module_ = module(frag); + + VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_}; + VkPushConstantRange push_constants{}; + push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; + push_constants.size = sizeof(PushConstants); + VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO}; + layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts; + layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants; + check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); + + get_pipeline(0, 2, 0); + create_offscreen_pass(); +} + +void VulkanWindow::create_offscreen_pass() { + VkFormatProperties properties{}; + vkGetPhysicalDeviceFormatProperties(physical_, VK_FORMAT_R5G6B5_UNORM_PACK16, &properties); + const VkFormatFeatureFlags wanted = + VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT; + offscreen_format_ = (properties.optimalTilingFeatures & wanted) == wanted + ? VK_FORMAT_R5G6B5_UNORM_PACK16 : VK_FORMAT_R8G8B8A8_UNORM; + VkAttachmentDescription attachments[2]{}; + attachments[0].format = offscreen_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[0].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; + attachments[0].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[0].initialLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + attachments[0].finalLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[1].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; + attachments[1].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[1].initialLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; + VkSubpassDescription subpass{}; + subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; + subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; + subpass.pDepthStencilAttachment = &depth_ref; + VkSubpassDependency dependencies[2]{}; + // Earlier work on the queue -- the previous frame sampling the texture, the previous + // offscreen pass, the first-use clear -- completes before this pass writes it. + dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL; dependencies[0].dstSubpass = 0; + dependencies[0].srcStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | + VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | + VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT; + dependencies[0].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | + VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT; + dependencies[0].dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | + VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT; + dependencies[0].dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | + VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; + // The back-buffer pass samples the result. + dependencies[1].srcSubpass = 0; dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL; + dependencies[1].srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; + dependencies[1].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; + dependencies[1].dstStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT; + dependencies[1].dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; + pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; + pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; + pass_info.dependencyCount = 2; pass_info.pDependencies = dependencies; + check(vkCreateRenderPass(device_, &pass_info, nullptr, &offscreen_pass_), "vkCreateRenderPass offscreen"); +} + +VkPipeline VulkanWindow::get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, + bool lines, std::uint8_t z_func, bool offscreen) { + const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | + (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | + (std::uint32_t(z_func & 0xfu) << 25) | (offscreen ? (1u << 29) : 0u); + if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; + + VkPipelineShaderStageCreateInfo stages[2]{}; + stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; + stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main"; + stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main"; + + VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX}; + VkVertexInputAttributeDescription attributes[9] = { + {0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)}, + {1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)}, + {2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)}, + {3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)}, + {4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)}, + {5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)}, + {6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)}, + {7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)}, + {8, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, vertex_fog)}}; + VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO}; + input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding; + input.vertexAttributeDescriptionCount = 9; input.pVertexAttributeDescriptions = attributes; + + VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO}; + assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; + + VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; + // An offscreen target is at most the device's 4096 (GetDeviceCaps); the viewport clips inside it. + VkRect2D scissor{{0, 0}, offscreen ? VkExtent2D{4096, 4096} : extent_}; + VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; + viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; + viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; + // D3DVIEWPORT8 changes per draw (SetViewport), so the viewport is dynamic state. + const VkDynamicState dynamic_states[] = {VK_DYNAMIC_STATE_VIEWPORT}; + VkPipelineDynamicStateCreateInfo dynamic_info{VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO}; + dynamic_info.dynamicStateCount = 1; dynamic_info.pDynamicStates = dynamic_states; + + VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO}; + raster.polygonMode = VK_POLYGON_MODE_FILL; + raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE); + raster.frontFace = VK_FRONT_FACE_CLOCKWISE; + raster.lineWidth = 1; + + VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; + multisample.rasterizationSamples = offscreen ? VK_SAMPLE_COUNT_1_BIT : samples_; + + VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; + depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; + depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE; + switch (z_func) { + case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break; + case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break; + case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break; + case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break; + case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break; + case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break; + case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break; + default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break; + } + + auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor { + switch (d3d) { + case 1: return VK_BLEND_FACTOR_ZERO; + case 2: return VK_BLEND_FACTOR_ONE; + case 3: return VK_BLEND_FACTOR_SRC_COLOR; + case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR; + case 5: return VK_BLEND_FACTOR_SRC_ALPHA; + case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + case 7: return VK_BLEND_FACTOR_DST_ALPHA; + case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; + case 9: return VK_BLEND_FACTOR_DST_COLOR; + case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR; + case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE; + default: return fallback; + } + }; + + VkPipelineColorBlendAttachmentState blend_attachment{}; + blend_attachment.colorWriteMask = 0xf; + if (blend > 0) { + blend_attachment.blendEnable = VK_TRUE; + if (blend >= 16) { + blend_attachment.srcColorBlendFactor = + d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA); + blend_attachment.dstColorBlendFactor = + d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA); + } else { + blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA; + blend_attachment.dstColorBlendFactor = + blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + } + blend_attachment.colorBlendOp = VK_BLEND_OP_ADD; + // D3D8 has no separate alpha blend: SRCBLEND/DESTBLEND apply to alpha as well. + auto alpha_factor = [](VkBlendFactor factor) { + switch (factor) { + case VK_BLEND_FACTOR_SRC_COLOR: return VK_BLEND_FACTOR_SRC_ALPHA; + case VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + case VK_BLEND_FACTOR_DST_COLOR: return VK_BLEND_FACTOR_DST_ALPHA; + case VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; + case VK_BLEND_FACTOR_SRC_ALPHA_SATURATE: return VK_BLEND_FACTOR_ONE; + default: return factor; + } + }; + blend_attachment.srcAlphaBlendFactor = alpha_factor(blend_attachment.srcColorBlendFactor); + blend_attachment.dstAlphaBlendFactor = alpha_factor(blend_attachment.dstColorBlendFactor); + blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD; + } + VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO}; + blending.attachmentCount = 1; blending.pAttachments = &blend_attachment; + + VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO}; + pipeline_info.stageCount = 2; pipeline_info.pStages = stages; + pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly; + pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster; + pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; + pipeline_info.pColorBlendState = &blending; + pipeline_info.pDynamicState = &dynamic_info; + pipeline_info.layout = layout_; pipeline_info.renderPass = offscreen ? offscreen_pass_ : pass_; + VkPipeline handle = VK_NULL_HANDLE; + check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); + pipelines_.emplace(key, handle); + return handle; +} + +} // namespace mt_host diff --git a/src/host/vk_resources.cpp b/src/host/vk_resources.cpp new file mode 100644 index 00000000..9054dcb6 --- /dev/null +++ b/src/host/vk_resources.cpp @@ -0,0 +1,453 @@ +// VulkanWindow: per-frame host buffers (bones, fixed-function state, UI, staging), descriptor pool, offscreen render targets and geometry buffers. +#include "vulkan_window.h" + +#include "host_util.h" + +#include +#include +#include + +namespace mt_host { + +void VulkanWindow::create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) { + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = size; + info.usage = usage; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host"); + VkMemoryRequirements reqs{}; + vkGetBufferMemoryRequirements(device_, buffer, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + alloc.memoryTypeIndex = find_memory_type( + reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host"); + check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host"); + check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host"); +} + +void VulkanWindow::create_descriptors_and_buffers() { + VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + sampler_info.magFilter = VK_FILTER_LINEAR; + sampler_info.minFilter = VK_FILTER_LINEAR; + sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; + sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.minLod = 0.0f; + sampler_info.maxLod = VK_LOD_CLAMP_NONE; + if (max_anisotropy_ > 1.0f) { + sampler_info.anisotropyEnable = VK_TRUE; + sampler_info.maxAnisotropy = max_anisotropy_; + } + check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler"); + + VkSamplerCreateInfo clamp_info = sampler_info; + clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp"); + + VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + ui_sampler_info.magFilter = VK_FILTER_LINEAR; + ui_sampler_info.minFilter = VK_FILTER_LINEAR; + ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST; + ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.minLod = 0.0f; + ui_sampler_info.maxLod = 0.0f; + ui_sampler_info.anisotropyEnable = VK_FALSE; + ui_sampler_info.maxAnisotropy = 1.0f; + check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui"); + + VkDescriptorSetLayoutBinding tex_bindings[2]{}; + tex_bindings[0].binding = 0; + tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + tex_bindings[0].descriptorCount = 1; + tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; + tex_bindings[1].binding = 1; + tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + tex_bindings[1].descriptorCount = 1; + tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; + VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; + layout_info.bindingCount = 2; + layout_info.pBindings = tex_bindings; + check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex"); + + VkDescriptorSetLayoutBinding bone_bindings[2]{}; + bone_bindings[0].binding = 0; + bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + bone_bindings[0].descriptorCount = 1; + bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT; + bone_bindings[1].binding = 1; + bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + bone_bindings[1].descriptorCount = 1; + bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; + VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; + bone_layout_info.bindingCount = 2; + bone_layout_info.pBindings = bone_bindings; + check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone"); + + VkDescriptorPoolSize pool_sizes[3] = { + {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4 * kMaxFramesInFlight}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4 * kMaxFramesInFlight}}; + VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; + pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; + pool_info.maxSets = 8192; + pool_info.poolSizeCount = 3; + pool_info.pPoolSizes = pool_sizes; + check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool"); + + const std::uint8_t white_pixel[4] = {255, 255, 255, 255}; + upload_texture(1, 1, white_pixel, fallback_texture_, false, false); + fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); + fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); + + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + const auto alignment = std::max( + 1, properties.limits.minStorageBufferOffsetAlignment); + state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; + staging_alignment_ = std::max(16, properties.limits.optimalBufferCopyOffsetAlignment); + + for (auto& slot : frames_) { + void* bone_raw = nullptr; + create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, slot.bone_buffer, slot.bone_memory, &bone_raw); + slot.bone_mapped = static_cast(bone_raw); + slot.bone_capacity = kMaxBonesPerFrame; + for (std::size_t i = 0; i < 16; ++i) slot.bone_mapped[i] = kIdentityMatrix[i]; + + slot.state_capacity = 256; + void* state_raw = nullptr; + create_host_buffer(slot.state_capacity * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + slot.state_buffer, slot.state_memory, &state_raw); + slot.state_mapped = static_cast(state_raw); + + VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + bone_alloc.descriptorPool = descriptor_pool_; + bone_alloc.descriptorSetCount = 1; + bone_alloc.pSetLayouts = &bone_descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &bone_alloc, &slot.bone_descriptor_set), "vkAllocateDescriptorSets bone"); + + VkDescriptorBufferInfo buffer_infos[2] = { + {slot.bone_buffer, 0, kBoneBufferBytes}, + {slot.state_buffer, 0, sizeof(FixedFunctionState)}}; + VkWriteDescriptorSet writes[2]{}; + for (int binding = 0; binding < 2; ++binding) { + writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[binding].dstSet = slot.bone_descriptor_set; + writes[binding].dstBinding = static_cast(binding); + writes[binding].descriptorCount = 1; + writes[binding].descriptorType = binding == 0 + ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + writes[binding].pBufferInfo = &buffer_infos[binding]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + + void* ui_raw = nullptr; + create_host_buffer( + kUiBufferBytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, + slot.ui_buffer, + slot.ui_memory, + &ui_raw); + slot.ui_mapped = static_cast(ui_raw); + slot.ui_vertex_capacity = kMaxUiVerticesPerFrame; + slot.ui_index_capacity = kMaxUiIndicesPerFrame; + slot.ui_vertex_bytes = kUiVertexBytes; + slot.staging_wanted = kStagingRingBytes; + } +} + +void VulkanWindow::release_retired_buffers(FrameSlot& slot) { + for (auto [buffer, memory] : slot.retired_buffers) { + vkDestroyBuffer(device_, buffer, nullptr); + vkFreeMemory(device_, memory, nullptr); + } + slot.retired_buffers.clear(); +} + +void VulkanWindow::reset_staging(FrameSlot& slot) { + release_retired_buffers(slot); + slot.staging_used = 0; + if (slot.staging_wanted <= slot.staging_capacity) return; + if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); + if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); + if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); + void* raw = nullptr; + slot.staging_capacity = std::min(slot.staging_wanted, kStagingRingMaxBytes); + create_host_buffer(slot.staging_capacity, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, + slot.staging_buffer, slot.staging_memory, &raw); + slot.staging_mapped = static_cast(raw); + slot.staging_wanted = slot.staging_capacity; +} + +void VulkanWindow::ensure_bone_capacity(std::size_t required) { + if (required <= f_->bone_capacity) return; + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); + if (required > max_bones || required > UINT32_MAX) + throw std::runtime_error("bone palette exceeds device storage-buffer limit"); + std::size_t next = std::min(max_bones, std::max(required, f_->bone_capacity * 2)); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); + vkUnmapMemory(device_, f_->bone_memory); + vkDestroyBuffer(device_, f_->bone_buffer, nullptr); + vkFreeMemory(device_, f_->bone_memory, nullptr); + f_->bone_buffer = buffer; + f_->bone_memory = memory; + f_->bone_mapped = static_cast(mapped); + f_->bone_capacity = next; + VkDescriptorBufferInfo info{f_->bone_buffer, 0, next * 16 * sizeof(float)}; + VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; + write.dstSet = f_->bone_descriptor_set; + write.dstBinding = 0; + write.descriptorCount = 1; + write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + write.pBufferInfo = &info; + vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); +} + +void VulkanWindow::ensure_state_capacity(std::size_t required) { + if (required <= f_->state_capacity) return; + const auto next = std::max(required, f_->state_capacity * 2); + if (next > UINT32_MAX / state_stride_) + throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + buffer, memory, &mapped); + vkUnmapMemory(device_, f_->state_memory); + vkDestroyBuffer(device_, f_->state_buffer, nullptr); + vkFreeMemory(device_, f_->state_memory, nullptr); + f_->state_buffer = buffer; + f_->state_memory = memory; + f_->state_mapped = static_cast(mapped); + f_->state_capacity = next; + VkDescriptorBufferInfo info{f_->state_buffer, 0, sizeof(FixedFunctionState)}; + VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; + write.dstSet = f_->bone_descriptor_set; + write.dstBinding = 1; + write.descriptorCount = 1; + write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + write.pBufferInfo = &info; + vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); +} + +void VulkanWindow::ensure_ui_capacity(std::size_t command_count) { + if (command_count <= f_->ui_vertex_capacity / 4 && command_count <= f_->ui_index_capacity / 6) return; + if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); + const std::size_t quads = std::max(command_count, f_->ui_vertex_capacity / 2); + const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); + const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); + if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) + throw std::runtime_error("UI geometry exceeds 128 MiB safety limit"); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(vertex_bytes + index_bytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); + vkUnmapMemory(device_, f_->ui_memory); + vkDestroyBuffer(device_, f_->ui_buffer, nullptr); + vkFreeMemory(device_, f_->ui_memory, nullptr); + f_->ui_buffer = buffer; + f_->ui_memory = memory; + f_->ui_mapped = static_cast(mapped); + f_->ui_vertex_capacity = quads * 4; + f_->ui_index_capacity = quads * 6; + f_->ui_vertex_bytes = vertex_bytes; +} + +VulkanWindow::RenderTarget* VulkanWindow::get_render_target(const std::string& name) { + if (auto it = render_targets_.find(name); it != render_targets_.end()) return &it->second; + unsigned id = 0, w = 0, h = 0; + if (std::sscanf(name.c_str(), "rt:%u:%ux%u", &id, &w, &h) != 3 || !w || !h || w > 4096 || h > 4096) + return nullptr; + RenderTarget& target = render_targets_[name]; + target.width = w; + target.height = h; + auto make_image = [&](VkFormat format, VkImageUsageFlags usage, VkImageAspectFlags aspect, + VkImage& image, VkDeviceMemory& memory, VkImageView& view) { + VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + info.imageType = VK_IMAGE_TYPE_2D; + info.format = format; + info.extent = {w, h, 1}; + info.mipLevels = 1; info.arrayLayers = 1; + info.samples = VK_SAMPLE_COUNT_1_BIT; + info.tiling = VK_IMAGE_TILING_OPTIMAL; + info.usage = usage; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &info, nullptr, &image), "vkCreateImage render target"); + VkMemoryRequirements reqs{}; + vkGetImageMemoryRequirements(device_, image, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory render target"); + check(vkBindImageMemory(device_, image, memory, 0), "vkBindImageMemory render target"); + VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view_info.image = image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = format; + view_info.subresourceRange = {aspect, 0, 1, 0, 1}; + check(vkCreateImageView(device_, &view_info, nullptr, &view), "vkCreateImageView render target"); + }; + make_image(offscreen_format_, + VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT, + VK_IMAGE_ASPECT_COLOR_BIT, target.texture.image, target.texture.memory, target.texture.view); + make_image(VK_FORMAT_D32_SFLOAT, + VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, + VK_IMAGE_ASPECT_DEPTH_BIT, target.depth_image, target.depth_memory, target.depth_view); + const VkImageView views[2] = {target.texture.view, target.depth_view}; + VkFramebufferCreateInfo fb{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + fb.renderPass = offscreen_pass_; + fb.attachmentCount = 2; fb.pAttachments = views; + fb.width = w; fb.height = h; fb.layers = 1; + check(vkCreateFramebuffer(device_, &fb, nullptr, &target.framebuffer), "vkCreateFramebuffer render target"); + target.texture.descriptor = allocate_texture_descriptor_set(target.texture.view, fallback_texture_.view); + target.texture.ui_descriptor = allocate_ui_texture_descriptor_set(target.texture.view, fallback_texture_.view); + return ⌖ +} + +void VulkanWindow::release_render_target(RenderTarget& target) { + if (target.framebuffer) vkDestroyFramebuffer(device_, target.framebuffer, nullptr); + if (target.depth_view) vkDestroyImageView(device_, target.depth_view, nullptr); + if (target.depth_image) vkDestroyImage(device_, target.depth_image, nullptr); + if (target.depth_memory) vkFreeMemory(device_, target.depth_memory, nullptr); + release_texture(target.texture); + target = {}; +} + +void VulkanWindow::initialize_render_target(RenderTarget& target) { + VkImageMemoryBarrier barriers[2]{}; + for (auto& b : barriers) { + b.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER; + b.srcQueueFamilyIndex = b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + b.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + b.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + b.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + } + barriers[0].image = target.texture.image; + barriers[0].subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + barriers[1].image = target.depth_image; + barriers[1].subresourceRange = {VK_IMAGE_ASPECT_DEPTH_BIT, 0, 1, 0, 1}; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 2, barriers); + const VkClearColorValue white{{1.0f, 1.0f, 1.0f, 1.0f}}; + const VkClearDepthStencilValue far{1.0f, 0}; + vkCmdClearColorImage(f_->command, target.texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &white, 1, + &barriers[0].subresourceRange); + vkCmdClearDepthStencilImage(f_->command, target.depth_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &far, 1, + &barriers[1].subresourceRange); + for (auto& b : barriers) { + b.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + b.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + } + barriers[0].newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barriers[0].dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_READ_BIT; + barriers[1].newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + barriers[1].dstAccessMask = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT; + vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | + VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, + 0, 0, nullptr, 0, nullptr, 2, barriers); + target.initialized = true; +} + +void VulkanWindow::release_geometry(Geometry& geometry) { + if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr); + if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr); + geometry = {}; +} + +void VulkanWindow::upload_geometry(const Render3DDraw& draw, Geometry& geometry) { + const ScopedMs timer(timings_.geometry_upload_ms); + if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) + throw std::runtime_error("invalid native geometry array size"); + const auto count = draw.positions.size() / 3; + if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX) + throw std::runtime_error("geometry exceeds Vulkan index range"); + // D3D8 feeds opaque white for a vertex format without D3DFVF_DIFFUSE. + const std::uint32_t default_argb = 0xffffffffu; + const bool has_skin = + draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4; + std::vector vertices(count); + for (std::size_t i = 0; i < count; ++i) { + vertices[i].position[0] = draw.positions[3*i]; + vertices[i].position[1] = draw.positions[3*i+1]; + vertices[i].position[2] = draw.positions[3*i+2]; + vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f; + vertices[i].vertex_fog = i < draw.vertex_fog.size() ? draw.vertex_fog[i] : 1.0f; + if (3*i + 2 < draw.normals.size()) { + vertices[i].normal[0] = draw.normals[3*i]; + vertices[i].normal[1] = draw.normals[3*i+1]; + vertices[i].normal[2] = draw.normals[3*i+2]; + } else { + vertices[i].normal[0] = 0.0f; + vertices[i].normal[1] = 0.0f; + vertices[i].normal[2] = 1.0f; + } + if (2*i + 1 < draw.uv0.size()) { + vertices[i].uv[0] = draw.uv0[2*i]; + vertices[i].uv[1] = draw.uv0[2*i+1]; + } else { + vertices[i].uv[0] = 0.0f; + vertices[i].uv[1] = 0.0f; + } + const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb; + vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f; + vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f; + vertices[i].color[2] = float(argb & 255) / 255.0f; + vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f; + if (has_skin) { + for (int k = 0; k < 4; ++k) { + vertices[i].joints[k] = draw.bone_indices[4*i + k]; + vertices[i].weights[k] = draw.bone_weights[4*i + k]; + } + } else { + vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0; + vertices[i].weights[0] = 1.0f; + vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f; + } + if (2*i + 1 < draw.uv1.size()) { + vertices[i].mask_uv[0] = draw.uv1[2*i]; + vertices[i].mask_uv[1] = draw.uv1[2*i+1]; + } else { + vertices[i].mask_uv[0] = 0.0f; + vertices[i].mask_uv[1] = 0.0f; + } + } + for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count"); + geometry.index_offset = vertices.size() * sizeof(Vertex); + const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t); + if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB"); + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type( + requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory"); + check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory"); + void* mapped = nullptr; + check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory"); + std::memcpy(mapped, vertices.data(), geometry.index_offset); + std::memcpy(static_cast(mapped) + geometry.index_offset, + draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); + vkUnmapMemory(device_, geometry.memory); + geometry.vertex_count = static_cast(count); + geometry.index_count = static_cast(draw.indices.size()); + ++upload_count_; + uploaded_bytes_ += bytes; +} + +} // namespace mt_host diff --git a/src/host/vk_textures.cpp b/src/host/vk_textures.cpp new file mode 100644 index 00000000..aad9b124 --- /dev/null +++ b/src/host/vk_textures.cpp @@ -0,0 +1,422 @@ +// VulkanWindow: texture decode/upload, samplers per D3D8 sampler state, texture descriptor sets and eviction. +#include "vulkan_window.h" + +#include "draw_capture.h" +#include "host_util.h" +#include "texture_decode.h" + +#include +#include +#include + +namespace mt_host { + +VkDescriptorSet VulkanWindow::allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, + VkSampler sampler0, + VkSampler sampler1) { + VkDescriptorSet set = VK_NULL_HANDLE; + VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + desc_alloc.descriptorPool = descriptor_pool_; + desc_alloc.descriptorSetCount = 1; + desc_alloc.pSetLayouts = &descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets"); + + VkDescriptorImageInfo image_infos[2]{}; + image_infos[0].sampler = sampler0 ? sampler0 : sampler_; + image_infos[0].imageView = view0; + image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_); + image_infos[1].imageView = view1; + image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + + VkWriteDescriptorSet writes[2]{}; + for (int b = 0; b < 2; ++b) { + writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[b].dstSet = set; + writes[b].dstBinding = static_cast(b); + writes[b].descriptorCount = 1; + writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + writes[b].pImageInfo = &image_infos[b]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + return set; +} + +VkDescriptorSet VulkanWindow::allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) { + VkDescriptorSet set = VK_NULL_HANDLE; + VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + desc_alloc.descriptorPool = descriptor_pool_; + desc_alloc.descriptorSetCount = 1; + desc_alloc.pSetLayouts = &descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui"); + + VkDescriptorImageInfo image_infos[2]{}; + image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_; + image_infos[0].imageView = view0; + image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_); + image_infos[1].imageView = view1; + image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + + VkWriteDescriptorSet writes[2]{}; + for (int b = 0; b < 2; ++b) { + writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[b].dstSet = set; + writes[b].dstBinding = static_cast(b); + writes[b].descriptorCount = 1; + writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + writes[b].pImageInfo = &image_infos[b]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + return set; +} + +VulkanWindow::GpuTexture& VulkanWindow::get_gpu_texture( + const std::string& texture_name, + const std::unordered_map>& capture_textures) { + if (texture_name.empty()) return fallback_texture_; + if (texture_name.rfind("rt:", 0) == 0) { + RenderTarget* target = get_render_target(texture_name); + return target ? target->texture : fallback_texture_; + } + auto [it, inserted] = textures_.try_emplace(texture_name); + it->second.last_used_frame = frame_number_; + if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) + return it->second; + if (inserted) { + it->second.descriptor = fallback_texture_.descriptor; + it->second.ui_descriptor = fallback_texture_.ui_descriptor; + } + it->second.last_decode_attempt_frame = frame_number_; + + mtimage::Image decoded; + if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) { + decoded = decode_texture_bytes(raw->second.data(), raw->second.size()); + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (!decoded.ok()) { + if (texture_name.rfind("mem:", 0) == 0) { + UIMemoryTexture mem_tex; + if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { + const auto mtra = native_draw_capture::encode_raw_argb_as_mtra( + static_cast(mem_tex.width), + static_cast(mem_tex.height), + mem_tex.argb.data()); + decoded = decode_texture_bytes(mtra.data(), mtra.size()); + } + } else { + std::vector pack_bytes; + if (read_live_pack_texture(texture_name, pack_bytes)) + decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size()); + } + } +#endif + if (!decoded.ok()) return it->second; + // 40250 CGraphicImageTexture: DXT DDS -> CreateDDSTexture with the file's levels; + // other files -> D3DXCreateTextureFromFileInMemoryEx(D3DX_DEFAULT) = full chain; + // memory images (mem:, CreateTexture(w, h, 1, ...)) -> one level. + const bool is_memory_tex = texture_name.rfind("mem:", 0) == 0; + last_texture_upload_name_ = texture_name; + upload_texture( + decoded.w, decoded.h, decoded.rgba.data(), it->second, true, + !is_memory_tex && decoded.file_mip_levels == 0, &decoded.mips); + it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view); + it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view); + return it->second; +} + +std::uint64_t VulkanWindow::sampler_state_key(const Render3DDraw& draw, int stage) const { + return (std::uint64_t(draw.address_u[stage] & 15u)) | + (std::uint64_t(draw.address_v[stage] & 15u) << 4) | + (std::uint64_t(draw.min_filter[stage] & 15u) << 8) | + (std::uint64_t(draw.mag_filter[stage] & 15u) << 12) | + (std::uint64_t(draw.mip_filter[stage] & 15u) << 16); +} + +VkSampler VulkanWindow::sampler_for_draw(const Render3DDraw& draw, int stage) { + const auto key = sampler_state_key(draw, stage); + if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_; + if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second; + auto address_mode = [](std::uint32_t mode) { + switch (mode) { + case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT; + case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER; + default: return VK_SAMPLER_ADDRESS_MODE_REPEAT; + } + }; + VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + info.addressModeU = address_mode(draw.address_u[stage]); + info.addressModeV = address_mode(draw.address_v[stage]); + info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; + info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE; + info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; + info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; + info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR; + info.minLod = 0.0f; + info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE; + if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) { + info.anisotropyEnable = VK_TRUE; + info.maxAnisotropy = max_anisotropy_; + } + VkSampler sampler = VK_NULL_HANDLE; + check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state"); + state_samplers_.emplace(key, sampler); + return sampler; +} + +VkDescriptorSet VulkanWindow::get_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures, + const Render3DDraw& draw) { + const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); + const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures); + const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" + + std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1)); + if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { + it->second.last_used_frame = frame_number_; + return it->second.set; + } + const VkDescriptorSet set = allocate_texture_descriptor_set( + tex0.view ? tex0.view : fallback_texture_.view, + tex1.view ? tex1.view : fallback_texture_.view, + sampler_for_draw(draw, 0), sampler_for_draw(draw, 1)); + paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); + return set; +} + +VkDescriptorSet VulkanWindow::get_ui_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures) { + const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); + if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; + const auto& tex1 = get_gpu_texture(mask_name, capture_textures); + if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; + const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name; + if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { + it->second.last_used_frame = frame_number_; + return it->second.set; + } + const VkDescriptorSet set = allocate_ui_texture_descriptor_set( + tex0.view ? tex0.view : fallback_texture_.view, + tex1.view ? tex1.view : fallback_texture_.view); + paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); + return set; +} + +void VulkanWindow::upload_texture( + std::uint32_t width, + std::uint32_t height, + const std::uint8_t* rgba, + GpuTexture& texture, + bool count_upload, + bool generate_mips, + const std::vector>* file_mips) { + const ScopedMs timer(timings_.texture_upload_ms); + const bool has_file_mips = !generate_mips && file_mips && !file_mips->empty(); + VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; + if (has_file_mips) + for (const auto& level : *file_mips) byte_size += level.size(); + const std::uint32_t mip_levels = generate_mips + ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) + : has_file_mips ? 1u + static_cast(file_mips->size()) : 1u; + // Inside a frame the copy is recorded ahead of that frame's render pass and reads the slot's + // staging ring; the GPU runs it with the frame, no queue wait. Outside a frame (the fallback + // texture at startup) it is a one-off submit that waits. + const bool in_frame = f_->recording; + VkBuffer staging_buffer = VK_NULL_HANDLE; + VkDeviceMemory staging_memory = VK_NULL_HANDLE; + VkDeviceSize staging_offset = 0; + std::uint8_t* mapped = nullptr; + const VkDeviceSize aligned_offset = + (f_->staging_used + staging_alignment_ - 1) / staging_alignment_ * staging_alignment_; + if (in_frame && f_->staging_mapped && aligned_offset + byte_size <= f_->staging_capacity) { + staging_buffer = f_->staging_buffer; + staging_offset = aligned_offset; + mapped = f_->staging_mapped + staging_offset; + f_->staging_used = aligned_offset + byte_size; + } else { + if (in_frame) f_->staging_wanted = std::max(f_->staging_wanted, aligned_offset + byte_size); + void* raw = nullptr; + create_host_buffer(byte_size, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, staging_buffer, staging_memory, &raw); + mapped = static_cast(raw); + } + std::memcpy(mapped, rgba, VkDeviceSize(width) * height * 4); + if (has_file_mips) { + auto* dst = mapped + VkDeviceSize(width) * height * 4; + for (const auto& level : *file_mips) { std::memcpy(dst, level.data(), level.size()); dst += level.size(); } + } + if (staging_memory) vkUnmapMemory(device_, staging_memory); + + VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + image_info.imageType = VK_IMAGE_TYPE_2D; + image_info.format = VK_FORMAT_R8G8B8A8_UNORM; + image_info.extent = {width, height, 1}; + image_info.mipLevels = mip_levels; + image_info.arrayLayers = 1; + image_info.samples = VK_SAMPLE_COUNT_1_BIT; + image_info.tiling = VK_IMAGE_TILING_OPTIMAL; + image_info.usage = + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT; + image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture"); + VkMemoryRequirements image_reqs{}; + vkGetImageMemoryRequirements(device_, texture.image, &image_reqs); + VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + image_alloc.allocationSize = image_reqs.size; + image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); + check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); + + VkCommandBuffer upload_cmd = f_->command; + if (!in_frame) { + VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); + } + + VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_transfer.image = texture.image; + to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &to_transfer); + + std::vector regions(has_file_mips ? mip_levels : 1u); + VkDeviceSize region_offset = 0; + for (std::uint32_t i = 0; i < regions.size(); ++i) { + const std::uint32_t lw = std::max(1u, width >> i), lh = std::max(1u, height >> i); + regions[i].bufferOffset = staging_offset + region_offset; + regions[i].imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; + regions[i].imageExtent = {lw, lh, 1}; + region_offset += VkDeviceSize(lw) * lh * 4; + } + vkCmdCopyBufferToImage( + upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + static_cast(regions.size()), regions.data()); + + std::int32_t mip_w = static_cast(width); + std::int32_t mip_h = static_cast(height); + for (std::uint32_t i = 1; generate_mips && i < mip_levels; ++i) { + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.image = texture.image; + barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1}; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &barrier); + + const std::int32_t next_w = std::max(1, mip_w / 2); + const std::int32_t next_h = std::max(1, mip_h / 2); + VkImageBlit blit{}; + blit.srcOffsets[0] = {0, 0, 0}; + blit.srcOffsets[1] = {mip_w, mip_h, 1}; + blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1}; + blit.dstOffsets[0] = {0, 0, 0}; + blit.dstOffsets[1] = {next_w, next_h, 1}; + blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; + vkCmdBlitImage( + upload_cmd, + texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + 1, &blit, VK_FILTER_LINEAR); + + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &barrier); + + mip_w = next_w; + mip_h = next_h; + } + + VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_shader.image = texture.image; + to_shader.subresourceRange = generate_mips + ? VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1} + : VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &to_shader); + + if (in_frame) { + if (staging_memory) f_->retired_buffers.emplace_back(staging_buffer, staging_memory); + } else { + check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; + check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); + check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); + vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); + vkDestroyBuffer(device_, staging_buffer, nullptr); + vkFreeMemory(device_, staging_memory, nullptr); + } + + VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; + view_info.format = VK_FORMAT_R8G8B8A8_UNORM; + view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture"); + + if (count_upload) { + ++texture_upload_count_; + texture_uploaded_bytes_ += byte_size; + } +} + +void VulkanWindow::release_texture(GpuTexture& texture) { + for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor}) + if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor) + vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set); + if (texture.view) vkDestroyImageView(device_, texture.view, nullptr); + if (texture.image) vkDestroyImage(device_, texture.image, nullptr); + if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr); + texture = {}; +} + +void VulkanWindow::prune_textures() { + constexpr std::uint64_t static_retention_frames = 120; + constexpr std::uint64_t memory_retention_frames = 2; + for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) { + const auto retention = it->first.find("mem:") != std::string::npos + ? memory_retention_frames : static_retention_frames; + if (frame_number_ - it->second.last_used_frame > retention) { + vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set); + it = paired_descriptors_.erase(it); + } else ++it; + } + for (auto it = textures_.begin(); it != textures_.end();) { + const auto retention = it->first.rfind("mem:", 0) == 0 + ? memory_retention_frames : static_retention_frames; + if (frame_number_ - it->second.last_used_frame > retention) { + release_texture(it->second); + it = textures_.erase(it); + } else ++it; + } +} + +} // namespace mt_host diff --git a/src/host/vulkan_window.cpp b/src/host/vulkan_window.cpp index 03a97a76..1b93111a 100644 --- a/src/host/vulkan_window.cpp +++ b/src/host/vulkan_window.cpp @@ -1,142 +1,19 @@ #include "vulkan_window.h" -#include "draw_capture.h" -#include "dxt.h" #include "host_util.h" -#include "input_keymap.h" -#include "texture_decode.h" - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT -#include "platform/ScriptLib/PythonBoot.h" -#endif #include -#ifdef __APPLE__ -#include -#endif #include #include +#include #include #include #include -#include -#include #include -#include namespace mt_host { -namespace { - -std::vector read_spirv(const char* name) { - std::size_t size = 0; - void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size); -#ifdef __ANDROID__ - if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size); - if (!bytes) bytes = SDL_LoadFile(name, &size); -#endif - if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name); - if (size == 0 || size % 4) { - SDL_free(bytes); - throw std::runtime_error(std::string("invalid SPIR-V size: ") + name); - } - std::vector words(size / 4); - std::memcpy(words.data(), bytes, size); - SDL_free(bytes); - return words; -} - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT -SDL_Cursor* load_game_cursor(int shape) { - static constexpr std::array images = { - "cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub", - "cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub", - "cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub", - "cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"}; - if (shape < 0 || shape >= static_cast(images.size())) return nullptr; - const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape]; - std::vector sub_bytes; - if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr; - std::unordered_map tokens; - std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end())); - std::string line; - while (std::getline(input, line)) { - std::istringstream fields(line); - std::string key, value; - if (!(fields >> key >> value)) continue; - if (!value.empty() && value.front() == '"') { - value.erase(0, 1); - if (!value.empty() && value.back() == '"') value.pop_back(); - } - std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); }); - tokens[key] = value; - } - if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr; - const std::string image_path = tokens["version"] == "2.0" - ? std::string("d:/ymir work/ui/cursor/") + tokens["image"] - : std::string("d:/ymir work/ui/") + tokens["image"]; - std::vector image_bytes; - if (!read_live_pack_texture(image_path, image_bytes)) return nullptr; - const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size()); - if (!image.ok()) return nullptr; - const int left = std::atoi(tokens["left"].c_str()); - const int top = std::atoi(tokens["top"].c_str()); - const int right = std::atoi(tokens["right"].c_str()); - const int bottom = std::atoi(tokens["bottom"].c_str()); - if (left < 0 || top < 0 || right <= left || bottom <= top || - right > image.w || bottom > image.h) return nullptr; - const int width = right - left, height = bottom - top; - std::vector pixels(std::size_t(width) * height * 4); - for (int y = 0; y < height; ++y) - std::memcpy(pixels.data() + std::size_t(y) * width * 4, - image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4, - std::size_t(width) * 4); - SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32, - pixels.data(), width * 4); - if (!surface) return nullptr; - const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0; - const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0; - SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y); - SDL_DestroySurface(surface); - return cursor; -} -#endif - -} // namespace - -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - -void VulkanWindow::enable_game_hardware_cursor() { - hardware_cursor_enabled_ = true; - for (int shape = 0; shape < static_cast(game_cursors_.size()); ++shape) { - game_cursors_[shape] = load_game_cursor(shape); - } - fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT); - SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_); - SDL_ShowCursor(); - os_cursor_hidden_ = false; -} - -void VulkanWindow::sync_game_cursor() { - if (!hardware_cursor_enabled_) return; - const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() && - !PythonBoot::IsSoftwareCursorVisible(); - if (visible == os_cursor_hidden_) { - if (visible) SDL_ShowCursor(); - else SDL_HideCursor(); - os_cursor_hidden_ = !visible; - } - if (!visible) return; - const int shape = PythonBoot::CursorShape(); - if (shape == current_cursor_shape_) return; - SDL_Cursor* cursor = shape >= 0 && shape < static_cast(game_cursors_.size()) - ? game_cursors_[shape] : nullptr; - if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_); - current_cursor_shape_ = shape; -} -#endif - VulkanWindow::VulkanWindow(bool vsync, int init_width, int init_height) : vsync_(vsync) { #ifdef __APPLE__ if (!std::getenv("VK_ICD_FILENAMES")) { @@ -274,254 +151,6 @@ VulkanWindow::~VulkanWindow() { SDL_Quit(); } -void VulkanWindow::write_bmp(const std::string& path, const std::uint8_t* pixels) const { - const std::uint32_t w = extent_.width, h = extent_.height; - const bool rgba = swapchain_format_ == VK_FORMAT_R8G8B8A8_UNORM || swapchain_format_ == VK_FORMAT_R8G8B8A8_SRGB; - std::vector file(54 + std::size_t(w) * h * 4); - auto put32 = [&](std::size_t at, std::uint32_t value) { std::memcpy(&file[at], &value, 4); }; - file[0] = 'B'; file[1] = 'M'; - put32(2, std::uint32_t(file.size())); put32(10, 54); put32(14, 40); - put32(18, w); put32(22, h); put32(26, 1u | (32u << 16)); put32(34, w * h * 4); - for (std::uint32_t y = 0; y < h; ++y) - for (std::uint32_t x = 0; x < w; ++x) { - const std::uint8_t* src = pixels + (std::size_t(y) * w + x) * 4; - std::uint8_t* dst = &file[54 + (std::size_t(h - 1 - y) * w + x) * 4]; - dst[0] = rgba ? src[2] : src[0]; dst[1] = src[1]; dst[2] = rgba ? src[0] : src[2]; dst[3] = 255; - } - std::ofstream out(path, std::ios::binary); - out.write(reinterpret_cast(file.data()), std::streamsize(file.size())); -} - -double VulkanWindow::input_idle_seconds() const { - if (fingers_down_ > 0 || buttons_down_ > 0) return 0.0; - return std::chrono::duration(std::chrono::steady_clock::now() - last_input_).count(); -} - -void VulkanWindow::request_screenshot(std::string path, std::uint64_t frame) { - screenshot_path_ = std::move(path); - screenshot_frame_ = frame; -} - -std::uint32_t VulkanWindow::logical_width() const { - int w = 0, h = 0; - if (window_) SDL_GetWindowSize(window_, &w, &h); -#ifdef __ANDROID__ - // Keep the 40250 UI at the same height as the macOS 1280x800 window, - // while retaining the device aspect ratio on wider phone displays. - if (w > 0 && h > 0) return std::max(1, int(std::lround(double(w) * 800.0 / h))); -#endif - return w > 0 ? static_cast(w) : extent_.width; -} - -std::uint32_t VulkanWindow::logical_height() const { - int w = 0, h = 0; - if (window_) SDL_GetWindowSize(window_, &w, &h); -#ifdef __ANDROID__ - if (w > 0 && h > 0) return 800; -#endif - return h > 0 ? static_cast(h) : extent_.height; -} - -VkViewport VulkanWindow::full_viewport() const { - return {0, 0, float(extent_.width), float(extent_.height), 0, 1}; -} - -int VulkanWindow::ui_safe_inset() const { - int pw = 0, ph = 0; - if (safe_inset_px_ <= 0 || !window_ || !SDL_GetWindowSizeInPixels(window_, &pw, &ph) || ph <= 0) return 0; - return int(std::lround(double(safe_inset_px_) * logical_height() / ph)); -} - -VkViewport VulkanWindow::draw_viewport(const Render3DDraw& draw) const { - if (draw.viewport[2] <= 0 || draw.viewport[3] <= 0) return full_viewport(); - const float sx = float(extent_.width) / float(logical_width()); - const float sy = float(extent_.height) / float(logical_height()); - return {(draw.viewport[0] + draw.ui_offset_x) * sx, draw.viewport[1] * sy, draw.viewport[2] * sx, draw.viewport[3] * sy, - draw.viewport_z[0], draw.viewport_z[1]}; -} - -void VulkanWindow::sync_screen_keyboard() { - const bool want = touch_controller.wants_screen_keyboard(); - const unsigned serial = touch_controller.screen_keyboard_serial(); - if (want && (!SDL_TextInputActive(window_) || serial != screen_keyboard_serial_)) { - SDL_PropertiesID props = SDL_CreateProperties(); - SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_TYPE_NUMBER, SDL_TEXTINPUT_TYPE_TEXT); - SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_CAPITALIZATION_NUMBER, SDL_CAPITALIZE_NONE); - SDL_SetBooleanProperty(props, SDL_PROP_TEXTINPUT_AUTOCORRECT_BOOLEAN, false); - SDL_StartTextInputWithProperties(window_, props); - SDL_DestroyProperties(props); - } else if (!want && SDL_TextInputActive(window_)) { - SDL_StopTextInput(window_); - } - screen_keyboard_serial_ = serial; -} - -bool VulkanWindow::poll(bool forward_to_live_client) { -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - auto map_mouse = [&](float wx, float wy, int& ox, int& oy) { - int win_w = 0, win_h = 0; - SDL_GetWindowSize(window_, &win_w, &win_h); - unsigned ui_w = extent_.width ? extent_.width : 960; - unsigned ui_h = extent_.height ? extent_.height : 640; - UIRenderGetSize(&ui_w, &ui_h); - if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) { - ox = static_cast(std::lround(wx * float(ui_w) / float(win_w))); - oy = static_cast(std::lround(wy * float(ui_h) / float(win_h))); - } else { - ox = static_cast(std::lround(wx)); - oy = static_cast(std::lround(wy)); - } - }; - std::optional> pending_mouse_move; - auto flush_mouse_move = [&] { - if (!pending_mouse_move) return; - const auto [mx, my] = *pending_mouse_move; - pending_mouse_move.reset(); - last_mx_ = mx; - last_my_ = my; - if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return; - PythonBoot::UIMouseMove(mx, my); - }; -#endif - SDL_Event event; - while (SDL_PollEvent(&event)) { - note_input(event); -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - // Touches arrive as SDL_EVENT_FINGER_*; the touch-synthesized mouse copy - // would register as a second finger (101) and turn every drag into a pinch. - if (touch_controller.is_enabled() && - ((event.type == SDL_EVENT_MOUSE_MOTION && event.motion.which == SDL_TOUCH_MOUSEID) || - ((event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) && - event.button.which == SDL_TOUCH_MOUSEID))) - continue; - if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) { - int mx = 0, my = 0; - map_mouse(event.motion.x, event.motion.y, mx, my); - pending_mouse_move = std::pair{mx, my}; - continue; - } - // Preserve event order at button/key/focus boundaries while collapsing - // high-frequency motion into the most recent position. - flush_mouse_move(); -#endif - if (event.type == SDL_EVENT_QUIT) return false; - if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) { - int ew = 0, eh = 0; - SDL_GetWindowSize(window_, &ew, &eh); - touch_controller.update_screen_size(int(logical_width()), int(logical_height()), ui_safe_inset()); -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (forward_to_live_client && ew > 0 && eh > 0) { - PythonBoot::SetUISafeInset(ui_safe_inset()); - PythonBoot::SetUISize(int(logical_width()), int(logical_height())); - } -#endif - recreate_swapchain(); - continue; - } - if (event.type == SDL_EVENT_FINGER_DOWN) { - touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); - continue; - } else if (event.type == SDL_EVENT_FINGER_MOTION) { - touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); - continue; - } else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) { - touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); - continue; - } -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (forward_to_live_client) { - if (!hardware_cursor_enabled_ && - (event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) { - std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n"); - SDL_HideCursor(); - os_cursor_hidden_ = true; - } else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) { - SDL_CaptureMouse(false); - PythonBoot::UIMouseButton(2, false, last_mx_, last_my_); - PythonBoot::UIMouseButton(3, false, last_mx_, last_my_); - PythonBoot::UIMouseButton(1, false, last_mx_, last_my_); - } - if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) { - const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN; - int btn = 0; - if (event.button.button == SDL_BUTTON_LEFT) btn = 1; - else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2; - else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3; - if (btn) { - int mx = 0, my = 0; - map_mouse(event.button.x, event.button.y, mx, my); - last_mx_ = mx; - last_my_ = my; - if (touch_controller.is_enabled()) { - if (touch_controller.on_mouse_button(btn, pressed, mx, my)) { - continue; - } - } - SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0); - PythonBoot::UIMouseButton(btn, pressed, mx, my); - } - } else if (event.type == SDL_EVENT_MOUSE_WHEEL) { - PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f)); - } else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) { - const bool pressed = event.type == SDL_EVENT_KEY_DOWN; - // Android back = ESC: closes the top window, or opens the system menu (game.py). - if (event.key.scancode == SDL_SCANCODE_AC_BACK) event.key.scancode = SDL_SCANCODE_ESCAPE; - if (pressed) { - if (const int vk = sdl_scancode_to_vk(event.key.scancode)) - PythonBoot::UIIMEKeyDown(vk); - if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8); - else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9); - else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER) - PythonBoot::UIChar(13); - else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27); - } - if (!event.key.repeat) { - if (const int dik = sdl_scancode_to_dik(event.key.scancode)) - PythonBoot::UIKey(dik, pressed); - } - } else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) { - const auto* s = reinterpret_cast(event.text.text); - while (*s) { - unsigned cp = 0; - if (*s < 0x80u) { - cp = *s++; - } else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) { - cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu); - s += 2; - } else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) { - cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu); - s += 3; - } else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) { - cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu); - s += 4; - } else { - ++s; - continue; - } - if (cp >= 32u && cp != 127u) - PythonBoot::UIChar(cp); - } - } - } -#else - (void)forward_to_live_client; -#endif - } -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - flush_mouse_move(); -#endif - touch_controller.update(); - if (touch_controller.is_enabled()) - sync_screen_keyboard(); -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) { - SDL_HideCursor(); - os_cursor_hidden_ = true; - } -#endif - return true; -} - void VulkanWindow::reset_timings() { finish_gpu_timings(); timings_ = {}; @@ -544,510 +173,6 @@ void VulkanWindow::set_frames_in_flight(std::uint32_t count) { frames_in_flight_ = std::clamp(count, 1, kMaxFramesInFlight); } -void VulkanWindow::render( - const std::vector& draws, - const std::unordered_map>& capture_textures, - std::uint32_t ui_width, - std::uint32_t ui_height, - const std::vector& ui_commands) { - if (extent_.width == 0 || extent_.height == 0) return; - const auto start = std::chrono::steady_clock::now(); - f_ = &frames_[frame_slot_ % frames_in_flight_]; - check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences"); - collect_pending_gpu_timestamp(*f_); - reset_staging(*f_); - const auto fenced = std::chrono::steady_clock::now(); - prune_textures(); - std::uint32_t image_index = 0; - const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, f_->acquire, VK_NULL_HANDLE, &image_index); - if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { - recreate_swapchain(); - return; - } - if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR"); - const auto synchronized = std::chrono::steady_clock::now(); - ++frame_number_; - ++timed_frames_; - while (rendered_.size() < swapchain_images_.size()) { - VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; - VkSemaphore semaphore = VK_NULL_HANDLE; - check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &semaphore), "vkCreateSemaphore rendered"); - rendered_.push_back(semaphore); - } - // Texture uploads of this frame are recorded ahead of the render pass (upload_texture). - check(vkResetCommandBuffer(f_->command, 0), "vkResetCommandBuffer"); - VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; - begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; - check(vkBeginCommandBuffer(f_->command, &begin), "vkBeginCommandBuffer"); - vkCmdResetQueryPool(f_->command, query_pool_, f_->query_base, 2); - vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, f_->query_base); - f_->recording = true; - - std::vector prepared_draws; - prepared_draws.reserve(draws.size()); - // Draws into offscreen render targets (the shadow map), in recorded order, drawn before the - // back-buffer pass that samples them. - std::vector offscreen_draws; - std::unordered_set active_keys; - std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; - std::uint32_t bone_cursor = 0; - std::size_t required_bones = 0; - for (const auto& draw : draws) { - if (!draw.bone_matrices.empty()) { - if (draw.bone_matrices.size() % 16) - throw std::runtime_error("invalid bone palette length"); - required_bones += draw.bone_matrices.size() / 16; - } - } - ensure_bone_capacity(required_bones); - - for (const auto& draw : draws) { - RenderTarget* target = nullptr; - if (draw.render_target) { - target = get_render_target( - Render3DRenderTargetName(draw.render_target, draw.target_width, draw.target_height)); - if (!target) continue; - } - if (draw.clear_flags && target) { - PreparedDraw clear{}; - clear.target = target; - clear.clear_flags = draw.clear_flags; - const auto color = unpack_argb(draw.clear_color); - clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; - clear.clear_depth.depthStencil = {draw.clear_z, 0}; - clear.clear_rect = {{0, 0}, {target->width, target->height}}; - offscreen_draws.push_back(clear); - continue; - } - if (draw.clear_flags) { - PreparedDraw clear{}; - clear.clear_flags = draw.clear_flags; - const auto color = unpack_argb(draw.clear_color); - clear.clear_color.color = {{color[0], color[1], color[2], color[3]}}; - clear.clear_depth.depthStencil = {draw.clear_z, 0}; - // D3D8 Clear with no rectangles clears the current viewport. - const auto viewport = draw_viewport(draw); - const float x0 = std::clamp(viewport.x, 0.0f, float(extent_.width)); - const float y0 = std::clamp(viewport.y, 0.0f, float(extent_.height)); - const float x1 = std::clamp(viewport.x + viewport.width, 0.0f, float(extent_.width)); - const float y1 = std::clamp(viewport.y + viewport.height, 0.0f, float(extent_.height)); - clear.clear_rect = {{std::int32_t(x0), std::int32_t(y0)}, - {std::uint32_t(x1 - x0), std::uint32_t(y1 - y0)}}; - prepared_draws.push_back(clear); - continue; - } - if (draw.positions.empty() || draw.indices.empty()) continue; - const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw); - const GeometryId id{draw.geometry_key, signature}; - auto [it, inserted] = geometries_.try_emplace(id); - auto& geometry = it->second; - if (inserted) { - upload_geometry(draw, geometry); - } else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) { - throw std::runtime_error("geometry key/revision reused with a different size"); - } - geometry.last_used_frame = frame_number_; - if (draw.geometry_key) active_keys.insert(draw.geometry_key); - - float skin_offset_encoded = 0.0f; - if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && f_->bone_mapped) { - const auto bone_count = static_cast(draw.bone_matrices.size() / 16); - if (std::size_t(bone_cursor) + bone_count <= f_->bone_capacity) { - std::memcpy( - f_->bone_mapped + std::size_t(bone_cursor) * 16, - draw.bone_matrices.data(), - draw.bone_matrices.size() * sizeof(float)); - skin_offset_encoded = float(bone_cursor + 1u); - bone_cursor += bone_count; - ++skinned_draw_count; - } else throw std::runtime_error("bone palette capacity exceeded"); - } - - const bool is_alpha = draw.alpha_blend != 0; - std::uint8_t cull = 0; - // D3D8 cull mode names describe the winding to discard. With the - // shader's Y flip, clockwise is Vulkan's front-facing winding. - if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise - else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise - const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0; - std::uint8_t blend = 0; - if (is_alpha) { - if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 && - draw.dest_blend >= 1 && draw.dest_blend <= 11) { - blend = static_cast((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4)); - } else { - blend = 1; - } - } - const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, - static_cast(draw.z_func), target != nullptr); - // D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples. - const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1; - const auto descriptor = get_texture_descriptor( - draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw); - - auto constants = make_push_constants(draw, skin_offset_encoded); - if (draw.pretransformed) { - const float screen_width = float(logical_width()); - const float screen_height = float(logical_height()); - // Screen pixels (y down) to D3D NDC (y up); the shader's Y flip then maps to Vulkan. - constants.mvp = {2.0f / screen_width, 0, 0, 0, - 0, -2.0f / screen_height, 0, 0, - 0, 0, 1, 0, - -1, 1, 0, 1}; - } - PreparedDraw prepared_draw{}; - prepared_draw.target = target; - prepared_draw.geometry = &geometry; - prepared_draw.pipeline = pipeline; - prepared_draw.descriptor = descriptor; - prepared_draw.constants = constants; - prepared_draw.fixed_state = make_fixed_function_state(draw); - // XYZRHW vertices are already in screen space and bypass the viewport transform. - prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw); - if (draw.pretransformed && draw.ui_offset_x != 0) - prepared_draw.viewport.x += draw.ui_offset_x * float(extent_.width) / float(logical_width()); - if (target) { - // D3DVIEWPORT8 in the target's own pixels. - prepared_draw.viewport = draw.viewport[2] > 0 && draw.viewport[3] > 0 - ? VkViewport{draw.viewport[0], draw.viewport[1], draw.viewport[2], draw.viewport[3], - draw.viewport_z[0], draw.viewport_z[1]} - : VkViewport{0, 0, float(target->width), float(target->height), 0, 1}; - offscreen_draws.push_back(prepared_draw); - } else prepared_draws.push_back(prepared_draw); - vertex_count += geometry.vertex_count; - index_count += geometry.index_count; - } - for (auto it = geometries_.begin(); it != geometries_.end();) { - const auto age = frame_number_ - it->second.last_used_frame; - const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key); - if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) { - release_geometry(it->second); - it = geometries_.erase(it); - } else ++it; - } - - std::vector ui_batches; - std::size_t ui_quad_count = 0; - if (f_->ui_mapped) { - const uint32_t target_ui_w = ui_width ? ui_width : 960; - const uint32_t target_ui_h = ui_height ? ui_height : 640; - update_fps_counter(); - if (touch_controller.is_enabled() || show_fps_) { - std::vector combined_ui = ui_commands; - if (touch_controller.is_enabled()) { - touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h), ui_safe_inset()); - touch_controller.append_ui_commands(combined_ui); - } - if (show_fps_ && fps_ >= 0) { - // Top left, inside the safe area and clear of 40250's corner icon. - const float inset = float(std::min(ui_safe_inset(), int(target_ui_w) / 8)); - TouchController::draw_segment_text(combined_ui, "FPS " + std::to_string(fps_), - inset + 40.0f, 8.0f, 12.0f, 0xFF40FF40); - } - if (!combined_ui.empty()) { - build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count); - } - } else if (!ui_commands.empty()) { - build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count); - } - } - - ensure_state_capacity(prepared_draws.size() + offscreen_draws.size() + ui_batches.size()); - std::size_t state_index = 0; - for (auto& draw : offscreen_draws) { - if (!draw.geometry) continue; - draw.state_offset = static_cast(state_index * state_stride_); - std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); - ++state_index; - } - last_offscreen_draw_count_ = offscreen_draws.size(); - for (auto& draw : prepared_draws) { - if (!draw.geometry) continue; - draw.state_offset = static_cast(state_index * state_stride_); - std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); - ++state_index; - } - const FixedFunctionState ui_state{}; - for (auto& batch : ui_batches) { - batch.state_offset = static_cast(state_index * state_stride_); - std::memcpy(f_->state_mapped + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); - ++state_index; - } - - const auto prepared = std::chrono::steady_clock::now(); - f_->recording = false; - check(vkResetFences(device_, 1, &f_->fence), "vkResetFences"); - - VkClearValue clears[3]{}; - // 40250 clears the back buffer to 0xff000000 once at device creation and afterwards - // only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer). - clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}}; - clears[1].depthStencil = {1.0f, 0}; - VkPipeline current_pipeline = VK_NULL_HANDLE; - VkDescriptorSet current_descriptor = VK_NULL_HANDLE; - - VkViewport current_viewport{-1, -1, -1, -1, -1, -1}; - auto set_viewport = [&](const VkViewport& viewport) { - if (std::memcmp(&viewport, ¤t_viewport, sizeof(viewport)) == 0) return; - vkCmdSetViewport(f_->command, 0, 1, &viewport); - current_viewport = viewport; - }; - - // One prepared draw or Clear inside the current render pass. - auto record_prepared = [&](const PreparedDraw& prepared_draw) { - if (!prepared_draw.geometry) { - VkClearAttachment attachments[2]{}; - std::uint32_t attachment_count = 0; - if (prepared_draw.clear_flags & 1u) { // D3DCLEAR_TARGET - attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; - attachments[attachment_count].colorAttachment = 0; - attachments[attachment_count].clearValue = prepared_draw.clear_color; - ++attachment_count; - } - if (prepared_draw.clear_flags & 2u) { // D3DCLEAR_ZBUFFER - attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; - attachments[attachment_count].clearValue = prepared_draw.clear_depth; - ++attachment_count; - } - const VkClearRect rect{prepared_draw.clear_rect, 0, 1}; - if (attachment_count && rect.rect.extent.width && rect.rect.extent.height) - vkCmdClearAttachments(f_->command, attachment_count, attachments, 1, &rect); - return; - } - set_viewport(prepared_draw.viewport); - if (prepared_draw.pipeline != current_pipeline) { - vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); - current_pipeline = prepared_draw.pipeline; - } - if (prepared_draw.descriptor != current_descriptor) { - vkCmdBindDescriptorSets( - f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); - current_descriptor = prepared_draw.descriptor; - } - vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, - &f_->bone_descriptor_set, 1, &prepared_draw.state_offset); - const auto& geometry = *prepared_draw.geometry; - const VkDeviceSize vertex_offset = 0; - vkCmdBindVertexBuffers(f_->command, 0, 1, &geometry.buffer, &vertex_offset); - vkCmdBindIndexBuffer(f_->command, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); - vkCmdPushConstants( - f_->command, - layout_, - VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, - 0, - sizeof(prepared_draw.constants), - &prepared_draw.constants); - vkCmdDrawIndexed(f_->command, geometry.index_count, 1, 0, 0, 0); - }; - - // Render targets first used this frame (drawn or only sampled) start cleared. - for (auto& entry : render_targets_) - if (!entry.second.initialized) initialize_render_target(entry.second); - // Offscreen passes: one per run of draws into the same target. - for (std::size_t i = 0; i < offscreen_draws.size();) { - RenderTarget* target = offscreen_draws[i].target; - VkRenderPassBeginInfo offscreen_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; - offscreen_begin.renderPass = offscreen_pass_; - offscreen_begin.framebuffer = target->framebuffer; - offscreen_begin.renderArea.extent = {target->width, target->height}; - vkCmdBeginRenderPass(f_->command, &offscreen_begin, VK_SUBPASS_CONTENTS_INLINE); - for (; i < offscreen_draws.size() && offscreen_draws[i].target == target; ++i) - record_prepared(offscreen_draws[i]); - vkCmdEndRenderPass(f_->command); - } - - VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; - pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); - pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears; - vkCmdBeginRenderPass(f_->command, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); - - auto record_ui_pass = [&](bool behind_3d) { - bool ui_bound = false; - set_viewport(full_viewport()); - for (const auto& batch : ui_batches) { - if (batch.behind_3d != behind_3d) continue; - if (!ui_bound) { - const VkDeviceSize v_offset = 0; - vkCmdBindVertexBuffers(f_->command, 0, 1, &f_->ui_buffer, &v_offset); - vkCmdBindIndexBuffer(f_->command, f_->ui_buffer, f_->ui_vertex_bytes, VK_INDEX_TYPE_UINT32); - ui_bound = true; - } - if (batch.pipeline != current_pipeline) { - vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); - current_pipeline = batch.pipeline; - } - if (batch.descriptor != current_descriptor) { - vkCmdBindDescriptorSets( - f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); - current_descriptor = batch.descriptor; - } - vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, - &f_->bone_descriptor_set, 1, &batch.state_offset); - vkCmdPushConstants( - f_->command, - layout_, - VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, - 0, - sizeof(batch.constants), - &batch.constants); - vkCmdDrawIndexed(f_->command, batch.index_count, 1, batch.first_index, 0, 0); - } - }; - - record_ui_pass(true); - - for (const auto& prepared_draw : prepared_draws) record_prepared(prepared_draw); - - record_ui_pass(false); - - vkCmdEndRenderPass(f_->command); - VkBuffer readback = VK_NULL_HANDLE; - VkDeviceMemory readback_memory = VK_NULL_HANDLE; - const bool screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_; - if (screenshot) { - const VkDeviceSize bytes = VkDeviceSize(extent_.width) * extent_.height * 4; - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = bytes; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; - check(vkCreateBuffer(device_, &info, nullptr, &readback), "vkCreateBuffer readback"); - VkMemoryRequirements requirements{}; - vkGetBufferMemoryRequirements(device_, readback, &requirements); - VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - allocation.allocationSize = requirements.size; - allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, - VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &allocation, nullptr, &readback_memory), "vkAllocateMemory readback"); - check(vkBindBufferMemory(device_, readback, readback_memory, 0), "vkBindBufferMemory readback"); - VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; - barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.oldLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.image = swapchain_images_.at(image_index); - barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, - VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - VkBufferImageCopy region{}; - region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; - region.imageExtent = {extent_.width, extent_.height, 1}; - vkCmdCopyImageToBuffer(f_->command, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, ®ion); - barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.dstAccessMask = 0; - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, - VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - } - // Debug: MT_DUMP_RT=prefix writes each render target of the screenshot frame to _.pgm. - struct RtDump { VkBuffer buffer; VkDeviceMemory memory; std::uint32_t w, h; }; - std::vector rt_dumps; - const char* dump_rt = std::getenv("MT_DUMP_RT"); - if (screenshot && dump_rt && *dump_rt) { - const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; - for (auto& entry : render_targets_) { - RenderTarget& rt = entry.second; - RtDump dump{VK_NULL_HANDLE, VK_NULL_HANDLE, rt.width, rt.height}; - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = VkDeviceSize(rt.width) * rt.height * texel; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT; - check(vkCreateBuffer(device_, &info, nullptr, &dump.buffer), "vkCreateBuffer rt dump"); - VkMemoryRequirements requirements{}; - vkGetBufferMemoryRequirements(device_, dump.buffer, &requirements); - VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - allocation.allocationSize = requirements.size; - allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits, - VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &allocation, nullptr, &dump.memory), "vkAllocateMemory rt dump"); - check(vkBindBufferMemory(device_, dump.buffer, dump.memory, 0), "vkBindBufferMemory rt dump"); - VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_SHADER_READ_BIT; - barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.image = rt.texture.image; - barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, - VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - VkBufferImageCopy region{}; - region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; - region.imageExtent = {rt.width, rt.height, 1}; - vkCmdCopyImageToBuffer(f_->command, rt.texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dump.buffer, 1, ®ion); - barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, - VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); - rt_dumps.push_back(dump); - } - } - vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, f_->query_base + 1); - check(vkEndCommandBuffer(f_->command), "vkEndCommandBuffer"); - f_->has_pending_query = true; - - const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; - VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; - submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &f_->acquire; submit.pWaitDstStageMask = &stage; - submit.commandBufferCount = 1; submit.pCommandBuffers = &f_->command; - submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_[image_index]; - check(vkQueueSubmit(queue_, 1, &submit, f_->fence), "vkQueueSubmit"); - ++frame_slot_; - if (screenshot) { - check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences readback"); - void* mapped = nullptr; - check(vkMapMemory(device_, readback_memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback"); - write_bmp(screenshot_path_, static_cast(mapped)); - vkUnmapMemory(device_, readback_memory); - const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4; - for (std::size_t n = 0; n < rt_dumps.size(); ++n) { - const auto& dump = rt_dumps[n]; - void* data = nullptr; - check(vkMapMemory(device_, dump.memory, 0, VK_WHOLE_SIZE, 0, &data), "vkMapMemory rt dump"); - const auto* bytes = static_cast(data); - std::ofstream out(std::string(dump_rt) + "_" + std::to_string(n) + ".pgm", std::ios::binary); - out << "P5\n" << dump.w << " " << dump.h << "\n255\n"; - for (std::size_t i = 0; i < std::size_t(dump.w) * dump.h; ++i) { - std::uint8_t g = texel == 2 ? std::uint8_t(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31) : bytes[i * 4]; - out.put(char(g)); - } - vkUnmapMemory(device_, dump.memory); - vkDestroyBuffer(device_, dump.buffer, nullptr); - vkFreeMemory(device_, dump.memory, nullptr); - } - vkDestroyBuffer(device_, readback, nullptr); - vkFreeMemory(device_, readback_memory, nullptr); - } - const auto submitted = std::chrono::steady_clock::now(); - VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; - present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_[image_index]; - present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; - const auto presented = vkQueuePresentKHR(queue_, &present); - // An Android surface can continuously report SUBOPTIMAL while the app deliberately - // uses the identity transform. The image is still presentable; rebuilding every frame - // stalls rendering without changing that status. - if (presented == VK_ERROR_OUT_OF_DATE_KHR) { - recreate_swapchain(); - } else if (presented != VK_SUCCESS && presented != VK_SUBOPTIMAL_KHR) { - check(presented, "vkQueuePresentKHR"); - } - const auto presented_at = std::chrono::steady_clock::now(); - const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); - last_sync_ms_ = std::chrono::duration(synchronized - start).count(); - timings_.sync_ms += last_sync_ms_; - timings_.fence_ms += std::chrono::duration(fenced - start).count(); - timings_.prepare_ms += prepare_ms; - if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; - timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); - timings_.present_ms += std::chrono::duration(presented_at - submitted).count(); - last_draw_count_ = prepared_draws.size(); - last_skinned_draw_count_ = skinned_draw_count; - last_ui_batch_count_ = ui_batches.size(); - last_ui_quad_count_ = ui_quad_count; - last_vertex_count_ = vertex_count; - last_index_count_ = index_count; -} - const char* VulkanWindow::present_mode_name() const { switch (present_mode_) { case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE"; @@ -1079,27 +204,6 @@ native_perf::RendererTotals VulkanWindow::perf_totals() const { return t; } -void VulkanWindow::recreate_swapchain() { - int w = 0, h = 0; - SDL_GetWindowSizeInPixels(window_, &w, &h); - if (w <= 0 || h <= 0) return; - vkDeviceWaitIdle(device_); - for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); - pipelines_.clear(); - for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr); - framebuffers_.clear(); - if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; } - if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; } - if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; } - destroy_msaa_color(); - for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); - image_views_.clear(); - if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; } - create_swapchain(); - create_framebuffers(); - get_pipeline(0, 2, 0); -} - void VulkanWindow::collect_pending_gpu_timestamp(FrameSlot& slot) { if (!slot.has_pending_query) return; slot.has_pending_query = false; @@ -1114,1533 +218,6 @@ void VulkanWindow::collect_pending_gpu_timestamp(FrameSlot& slot) { } } -void VulkanWindow::build_ui_batches( - std::uint32_t ui_width, - std::uint32_t ui_height, - const std::vector& commands, - const std::unordered_map>& capture_textures, - std::vector& batches, - std::size_t& out_quad_count) { - ensure_ui_capacity(commands.size()); - auto* vertices = reinterpret_cast(f_->ui_mapped); - auto* indices = reinterpret_cast(f_->ui_mapped + f_->ui_vertex_bytes); - std::uint32_t v_count = 0; - std::uint32_t i_count = 0; - const float inv_w = 2.0f / float(ui_width); - const float inv_h = 2.0f / float(ui_height); - - for (const auto& cmd : commands) { - if (v_count + 4 > f_->ui_vertex_capacity || i_count + 6 > f_->ui_index_capacity) - throw std::runtime_error("UI command buffer capacity exceeded"); - if (cmd.kind == UIRenderCommand::Text) continue; - - float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2; - float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev; - const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1; - - float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; - std::array top_color = unpack_argb(cmd.argb); - std::array bot_color = - cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color; - if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue; - - if (cmd.kind == UIRenderCommand::Line) { - const float dx = x2 - x1, dy = y2 - y1; - const float len = std::sqrt(dx * dx + dy * dy); - if (len < 0.1f) continue; - const float nx = -dy / len * 0.5f; - const float ny = dx / len * 0.5f; - qx[0] = x1 + nx; qy[0] = y1 + ny; - qx[1] = x2 + nx; qy[1] = y2 + ny; - qx[2] = x1 - nx; qy[2] = y1 - ny; - qx[3] = x2 - nx; qy[3] = y2 - ny; - for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; } - } else if (cmd.kind == UIRenderCommand::Image && cmd.quad) { - const bool axis_aligned = - std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f && - std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f && - std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f && - std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f; - for (int k = 0; k < 4; ++k) { - mu[k] = cmd.mu[k]; - mv[k] = cmd.mv[k]; - } - if (has_clip && axis_aligned) { - const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3]; - const float cx1 = std::max(ox1, cmd.clip_x1); - const float cy1 = std::max(oy1, cmd.clip_y1); - const float cx2 = std::min(ox2, cmd.clip_x2); - const float cy2 = std::min(oy2, cmd.clip_y2); - if (cx2 <= cx1 || cy2 <= cy1) continue; - float msu = cmd.mu[0], meu = cmd.mu[1]; - float msv = cmd.mv[0], mev = cmd.mv[2]; - if (ox2 > ox1) { - const float t0 = (cx1 - ox1) / (ox2 - ox1); - const float t1 = (cx2 - ox1) / (ox2 - ox1); - const float du = eu - su; - su = cmd.su + du * t0; - eu = cmd.su + du * t1; - const float mdu = meu - msu; - msu = cmd.mu[0] + mdu * t0; - meu = cmd.mu[0] + mdu * t1; - } - if (oy2 > oy1) { - const float t0 = (cy1 - oy1) / (oy2 - oy1); - const float t1 = (cy2 - oy1) / (oy2 - oy1); - const float dv = ev - sv; - sv = cmd.sv + dv * t0; - ev = cmd.sv + dv * t1; - const float mdv = mev - msv; - msv = cmd.mv[0] + mdv * t0; - mev = cmd.mv[0] + mdv * t1; - } - qx[0] = cx1; qy[0] = cy1; - qx[1] = cx2; qy[1] = cy1; - qx[2] = cx1; qy[2] = cy2; - qx[3] = cx2; qy[3] = cy2; - mu[0] = msu; mv[0] = msv; - mu[1] = meu; mv[1] = msv; - mu[2] = msu; mv[2] = mev; - mu[3] = meu; mv[3] = mev; - } else { - if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) - continue; - for (int k = 0; k < 4; ++k) { - qx[k] = cmd.qx[k]; - qy[k] = cmd.qy[k]; - } - } - u[0] = su; v[0] = sv; - u[1] = eu; v[1] = sv; - u[2] = su; v[2] = ev; - u[3] = eu; v[3] = ev; - } else if (cmd.kind == UIRenderCommand::Bar && cmd.quad) { - // An untextured polygon piece (CPythonGraphic::RenderCoolTimeBox's fan triangles). - if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) - continue; - for (int k = 0; k < 4; ++k) { - qx[k] = cmd.qx[k]; - qy[k] = cmd.qy[k]; - u[k] = 0.0f; - v[k] = 0.0f; - } - } else { - if (has_clip) { - x1 = std::max(x1, cmd.clip_x1); - y1 = std::max(y1, cmd.clip_y1); - x2 = std::min(x2, cmd.clip_x2); - y2 = std::min(y2, cmd.clip_y2); - } - if (x2 <= x1 || y2 <= y1) continue; - qx[0] = x1; qy[0] = y1; - qx[1] = x2; qy[1] = y1; - qx[2] = x1; qy[2] = y2; - qx[3] = x2; qy[3] = y2; - u[0] = su; v[0] = sv; - u[1] = eu; v[1] = sv; - u[2] = su; v[2] = ev; - u[3] = eu; v[3] = ev; - } - - const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string(); - const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string(); - const bool is_masked = !mask_name.empty(); - // CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use - // INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR. - const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image - ? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1)) - : 1; - const VkPipeline pipeline = get_pipeline(0, 0, blend_mode); - const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures); - - for (int k = 0; k < 4; ++k) { - Vertex& vert = vertices[v_count + k]; - vert.position[0] = qx[k] * inv_w - 1.0f; - vert.position[1] = 1.0f - qy[k] * inv_h; - vert.position[2] = 0.0f; - vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f; - vert.uv[0] = u[k]; vert.uv[1] = v[k]; - const auto& col = (k < 2) ? top_color : bot_color; - vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3]; - vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0; - vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f; - vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k]; - vert.rhw = 1.0f; - } - indices[i_count + 0] = v_count + 0; - indices[i_count + 1] = v_count + 1; - indices[i_count + 2] = v_count + 2; - indices[i_count + 3] = v_count + 2; - indices[i_count + 4] = v_count + 1; - indices[i_count + 3 + 2] = v_count + 3; - - const float mask_flag = is_masked ? 1.0f : 0.0f; - if (!batches.empty() && - batches.back().behind_3d == cmd.behind_3d && - batches.back().pipeline == pipeline && - batches.back().descriptor == descriptor && - batches.back().constants.params[2] == mask_flag) { - batches.back().index_count += 6; - } else { - UiBatch batch{}; - batch.first_index = i_count; - batch.index_count = 6; - batch.pipeline = pipeline; - batch.descriptor = descriptor; - batch.behind_3d = cmd.behind_3d; - batch.constants.mvp = kIdentityMatrix; - batch.constants.params = {0.0f, 0.0f, mask_flag, 0.0f}; - batches.push_back(batch); - } - v_count += 4; - i_count += 6; - ++out_quad_count; - } -} - -std::uint32_t VulkanWindow::find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const { - VkPhysicalDeviceMemoryProperties properties{}; - vkGetPhysicalDeviceMemoryProperties(physical_, &properties); - for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i) - if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags) - return i; - throw std::runtime_error("no matching Vulkan memory type"); -} - -void VulkanWindow::select_device() { - std::uint32_t count = 0; - check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count"); - if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK"); - std::vector candidates(count); - check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices"); - for (auto candidate : candidates) { - std::uint32_t family_count = 0; - vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr); - std::vector families(family_count); - vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data()); - for (std::uint32_t i = 0; i < family_count; ++i) { - VkBool32 present = VK_FALSE; - check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support"); - if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) { - physical_ = candidate; queue_family_ = i; break; - } - } - if (physical_) break; - } - if (!physical_) throw std::runtime_error("no graphics/present queue family"); - VkPhysicalDeviceProperties properties{}; - vkGetPhysicalDeviceProperties(physical_, &properties); - device_name_ = properties.deviceName; - timestamp_period_ns_ = properties.limits.timestampPeriod; - VkPhysicalDeviceFeatures supported_features{}; - vkGetPhysicalDeviceFeatures(physical_, &supported_features); - VkPhysicalDeviceFeatures enabled_features{}; - if (supported_features.samplerAnisotropy) { - enabled_features.samplerAnisotropy = VK_TRUE; - max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy); - } - // Multisampled 3D edges; MT_MSAA=1 keeps the single-sample path, 2/4/8 request a count. - { - int wanted = 4; - if (const char* env = std::getenv("MT_MSAA")) wanted = std::max(1, std::atoi(env)); - const VkSampleCountFlags supported = - properties.limits.framebufferColorSampleCounts & properties.limits.framebufferDepthSampleCounts; - samples_ = VK_SAMPLE_COUNT_1_BIT; - for (VkSampleCountFlagBits c : {VK_SAMPLE_COUNT_8_BIT, VK_SAMPLE_COUNT_4_BIT, VK_SAMPLE_COUNT_2_BIT}) - if (int(c) <= wanted && (supported & c)) { samples_ = c; break; } - } - std::uint32_t extension_count = 0; - check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count"); - std::vector available(extension_count); - check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions"); - std::vector extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME}; - constexpr const char* portability_subset = "VK_KHR_portability_subset"; - for (const auto& extension : available) - if (std::strcmp(extension.extensionName, portability_subset) == 0) - extensions.push_back(portability_subset); - const float priority = 1.0f; - VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; - queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority; - VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; - device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info; - device_info.enabledExtensionCount = static_cast(extensions.size()); - device_info.ppEnabledExtensionNames = extensions.data(); - device_info.pEnabledFeatures = &enabled_features; - check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice"); - vkGetDeviceQueue(device_, queue_family_, 0, &queue_); -} - -void VulkanWindow::create_swapchain() { - VkSurfaceCapabilitiesKHR capabilities{}; - check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities"); - std::uint32_t count = 0; - check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count"); - if (!count) throw std::runtime_error("surface has no formats"); - std::vector formats(count); - check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats"); - VkSurfaceFormatKHR format = formats.front(); - for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate; - swapchain_format_ = format.format; - int width = 0, height = 0; - SDL_GetWindowSizeInPixels(window_, &width, &height); - if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent; - else { - extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width); - extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height); - } - - present_mode_ = VK_PRESENT_MODE_FIFO_KHR; - if (!vsync_) { - std::uint32_t mode_count = 0; - check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count"); - std::vector modes(mode_count); - if (mode_count) - check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes"); - if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end()) - present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR; - else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end()) - present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR; - } - - VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR}; - info.surface = surface_; - info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX); - info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_; - info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | - (capabilities.supportedUsageFlags & VK_IMAGE_USAGE_TRANSFER_SRC_BIT); - info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; - // Android surfaces can report a quarter-turn transform even when SDL's window is - // already landscape. Rendering in SDL window coordinates and requesting that transform - // again rotates the entire game image on presentation. - info.preTransform = (capabilities.supportedTransforms & VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR) - ? VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR : capabilities.currentTransform; -#ifdef __ANDROID__ - SDL_Log("swapchain: window=%dx%d extent=%ux%u currentTransform=%u preTransform=%u", - width, height, extent_.width, extent_.height, - unsigned(capabilities.currentTransform), unsigned(info.preTransform)); -#endif - info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_; - info.clipped = VK_TRUE; - check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR"); - check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count"); - std::vector images(count); - check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images"); - swapchain_images_ = images; - for (auto image : images) { - VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; - view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; - view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; - VkImageView handle = VK_NULL_HANDLE; - check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView"); - image_views_.push_back(handle); - } - VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - depth_info.imageType = VK_IMAGE_TYPE_2D; - depth_info.format = VK_FORMAT_D32_SFLOAT; - depth_info.extent = {extent_.width, extent_.height, 1}; - depth_info.mipLevels = 1; depth_info.arrayLayers = 1; - depth_info.samples = samples_; - depth_info.tiling = VK_IMAGE_TILING_OPTIMAL; - depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT; - depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth"); - VkMemoryRequirements depth_reqs{}; - vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs); - VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - depth_alloc.allocationSize = depth_reqs.size; - depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth"); - check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth"); - VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D; - depth_view.format = VK_FORMAT_D32_SFLOAT; - depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; - depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1; - check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth"); - if (samples_ != VK_SAMPLE_COUNT_1_BIT) create_msaa_color(); -} - -void VulkanWindow::create_msaa_color() { - VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - info.imageType = VK_IMAGE_TYPE_2D; - info.format = swapchain_format_; - info.extent = {extent_.width, extent_.height, 1}; - info.mipLevels = 1; info.arrayLayers = 1; - info.samples = samples_; - info.tiling = VK_IMAGE_TILING_OPTIMAL; - info.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSIENT_ATTACHMENT_BIT; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &info, nullptr, &msaa_image_), "vkCreateImage msaa color"); - VkMemoryRequirements reqs{}; - vkGetImageMemoryRequirements(device_, msaa_image_, &reqs); - VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - alloc.allocationSize = reqs.size; - try { - alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_LAZILY_ALLOCATED_BIT); - } catch (const std::runtime_error&) { - alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - } - check(vkAllocateMemory(device_, &alloc, nullptr, &msaa_memory_), "vkAllocateMemory msaa color"); - check(vkBindImageMemory(device_, msaa_image_, msaa_memory_, 0), "vkBindImageMemory msaa color"); - VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view.image = msaa_image_; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; - view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; - view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; - check(vkCreateImageView(device_, &view, nullptr, &msaa_view_), "vkCreateImageView msaa color"); -} - -void VulkanWindow::destroy_msaa_color() { - if (msaa_view_) { vkDestroyImageView(device_, msaa_view_, nullptr); msaa_view_ = VK_NULL_HANDLE; } - if (msaa_image_) { vkDestroyImage(device_, msaa_image_, nullptr); msaa_image_ = VK_NULL_HANDLE; } - if (msaa_memory_) { vkFreeMemory(device_, msaa_memory_, nullptr); msaa_memory_ = VK_NULL_HANDLE; } -} - -void VulkanWindow::create_framebuffers() { - for (auto view : image_views_) { - // Single-sample: {swapchain, depth}. MSAA: {msaa colour, depth, swapchain resolve}. - VkImageView fb_attachments[3] = {view, depth_view_, VK_NULL_HANDLE}; - std::uint32_t count = 2; - if (samples_ != VK_SAMPLE_COUNT_1_BIT) { - fb_attachments[0] = msaa_view_; - fb_attachments[2] = view; - count = 3; - } - VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; - framebuffer.renderPass = pass_; framebuffer.attachmentCount = count; framebuffer.pAttachments = fb_attachments; - framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; - VkFramebuffer handle = VK_NULL_HANDLE; - check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer"); - framebuffers_.push_back(handle); - } -} - -void VulkanWindow::create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) { - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = size; - info.usage = usage; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host"); - VkMemoryRequirements reqs{}; - vkGetBufferMemoryRequirements(device_, buffer, &reqs); - VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - alloc.allocationSize = reqs.size; - alloc.memoryTypeIndex = find_memory_type( - reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host"); - check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host"); - check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host"); -} - -void VulkanWindow::create_descriptors_and_buffers() { - VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; - sampler_info.magFilter = VK_FILTER_LINEAR; - sampler_info.minFilter = VK_FILTER_LINEAR; - sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; - sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT; - sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT; - sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; - sampler_info.minLod = 0.0f; - sampler_info.maxLod = VK_LOD_CLAMP_NONE; - if (max_anisotropy_ > 1.0f) { - sampler_info.anisotropyEnable = VK_TRUE; - sampler_info.maxAnisotropy = max_anisotropy_; - } - check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler"); - - VkSamplerCreateInfo clamp_info = sampler_info; - clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp"); - - VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; - ui_sampler_info.magFilter = VK_FILTER_LINEAR; - ui_sampler_info.minFilter = VK_FILTER_LINEAR; - ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST; - ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - ui_sampler_info.minLod = 0.0f; - ui_sampler_info.maxLod = 0.0f; - ui_sampler_info.anisotropyEnable = VK_FALSE; - ui_sampler_info.maxAnisotropy = 1.0f; - check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui"); - - VkDescriptorSetLayoutBinding tex_bindings[2]{}; - tex_bindings[0].binding = 0; - tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - tex_bindings[0].descriptorCount = 1; - tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; - tex_bindings[1].binding = 1; - tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - tex_bindings[1].descriptorCount = 1; - tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; - VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; - layout_info.bindingCount = 2; - layout_info.pBindings = tex_bindings; - check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex"); - - VkDescriptorSetLayoutBinding bone_bindings[2]{}; - bone_bindings[0].binding = 0; - bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; - bone_bindings[0].descriptorCount = 1; - bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT; - bone_bindings[1].binding = 1; - bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - bone_bindings[1].descriptorCount = 1; - bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; - VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; - bone_layout_info.bindingCount = 2; - bone_layout_info.pBindings = bone_bindings; - check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone"); - - VkDescriptorPoolSize pool_sizes[3] = { - {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, - {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4 * kMaxFramesInFlight}, - {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4 * kMaxFramesInFlight}}; - VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; - pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; - pool_info.maxSets = 8192; - pool_info.poolSizeCount = 3; - pool_info.pPoolSizes = pool_sizes; - check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool"); - - const std::uint8_t white_pixel[4] = {255, 255, 255, 255}; - upload_texture(1, 1, white_pixel, fallback_texture_, false, false); - fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); - fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); - - VkPhysicalDeviceProperties properties{}; - vkGetPhysicalDeviceProperties(physical_, &properties); - const auto alignment = std::max( - 1, properties.limits.minStorageBufferOffsetAlignment); - state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; - staging_alignment_ = std::max(16, properties.limits.optimalBufferCopyOffsetAlignment); - - for (auto& slot : frames_) { - void* bone_raw = nullptr; - create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, slot.bone_buffer, slot.bone_memory, &bone_raw); - slot.bone_mapped = static_cast(bone_raw); - slot.bone_capacity = kMaxBonesPerFrame; - for (std::size_t i = 0; i < 16; ++i) slot.bone_mapped[i] = kIdentityMatrix[i]; - - slot.state_capacity = 256; - void* state_raw = nullptr; - create_host_buffer(slot.state_capacity * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, - slot.state_buffer, slot.state_memory, &state_raw); - slot.state_mapped = static_cast(state_raw); - - VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - bone_alloc.descriptorPool = descriptor_pool_; - bone_alloc.descriptorSetCount = 1; - bone_alloc.pSetLayouts = &bone_descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &bone_alloc, &slot.bone_descriptor_set), "vkAllocateDescriptorSets bone"); - - VkDescriptorBufferInfo buffer_infos[2] = { - {slot.bone_buffer, 0, kBoneBufferBytes}, - {slot.state_buffer, 0, sizeof(FixedFunctionState)}}; - VkWriteDescriptorSet writes[2]{}; - for (int binding = 0; binding < 2; ++binding) { - writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[binding].dstSet = slot.bone_descriptor_set; - writes[binding].dstBinding = static_cast(binding); - writes[binding].descriptorCount = 1; - writes[binding].descriptorType = binding == 0 - ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - writes[binding].pBufferInfo = &buffer_infos[binding]; - } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); - - void* ui_raw = nullptr; - create_host_buffer( - kUiBufferBytes, - VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, - slot.ui_buffer, - slot.ui_memory, - &ui_raw); - slot.ui_mapped = static_cast(ui_raw); - slot.ui_vertex_capacity = kMaxUiVerticesPerFrame; - slot.ui_index_capacity = kMaxUiIndicesPerFrame; - slot.ui_vertex_bytes = kUiVertexBytes; - slot.staging_wanted = kStagingRingBytes; - } -} - -void VulkanWindow::release_retired_buffers(FrameSlot& slot) { - for (auto [buffer, memory] : slot.retired_buffers) { - vkDestroyBuffer(device_, buffer, nullptr); - vkFreeMemory(device_, memory, nullptr); - } - slot.retired_buffers.clear(); -} - -void VulkanWindow::reset_staging(FrameSlot& slot) { - release_retired_buffers(slot); - slot.staging_used = 0; - if (slot.staging_wanted <= slot.staging_capacity) return; - if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory); - if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr); - if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr); - void* raw = nullptr; - slot.staging_capacity = std::min(slot.staging_wanted, kStagingRingMaxBytes); - create_host_buffer(slot.staging_capacity, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, - slot.staging_buffer, slot.staging_memory, &raw); - slot.staging_mapped = static_cast(raw); - slot.staging_wanted = slot.staging_capacity; -} - -void VulkanWindow::ensure_bone_capacity(std::size_t required) { - if (required <= f_->bone_capacity) return; - VkPhysicalDeviceProperties properties{}; - vkGetPhysicalDeviceProperties(physical_, &properties); - const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); - if (required > max_bones || required > UINT32_MAX) - throw std::runtime_error("bone palette exceeds device storage-buffer limit"); - std::size_t next = std::min(max_bones, std::max(required, f_->bone_capacity * 2)); - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - void* mapped = nullptr; - create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, f_->bone_memory); - vkDestroyBuffer(device_, f_->bone_buffer, nullptr); - vkFreeMemory(device_, f_->bone_memory, nullptr); - f_->bone_buffer = buffer; - f_->bone_memory = memory; - f_->bone_mapped = static_cast(mapped); - f_->bone_capacity = next; - VkDescriptorBufferInfo info{f_->bone_buffer, 0, next * 16 * sizeof(float)}; - VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; - write.dstSet = f_->bone_descriptor_set; - write.dstBinding = 0; - write.descriptorCount = 1; - write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; - write.pBufferInfo = &info; - vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); -} - -void VulkanWindow::ensure_state_capacity(std::size_t required) { - if (required <= f_->state_capacity) return; - const auto next = std::max(required, f_->state_capacity * 2); - if (next > UINT32_MAX / state_stride_) - throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - void* mapped = nullptr; - create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, - buffer, memory, &mapped); - vkUnmapMemory(device_, f_->state_memory); - vkDestroyBuffer(device_, f_->state_buffer, nullptr); - vkFreeMemory(device_, f_->state_memory, nullptr); - f_->state_buffer = buffer; - f_->state_memory = memory; - f_->state_mapped = static_cast(mapped); - f_->state_capacity = next; - VkDescriptorBufferInfo info{f_->state_buffer, 0, sizeof(FixedFunctionState)}; - VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; - write.dstSet = f_->bone_descriptor_set; - write.dstBinding = 1; - write.descriptorCount = 1; - write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; - write.pBufferInfo = &info; - vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); -} - -void VulkanWindow::ensure_ui_capacity(std::size_t command_count) { - if (command_count <= f_->ui_vertex_capacity / 4 && command_count <= f_->ui_index_capacity / 6) return; - if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); - const std::size_t quads = std::max(command_count, f_->ui_vertex_capacity / 2); - const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); - const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); - if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) - throw std::runtime_error("UI geometry exceeds 128 MiB safety limit"); - VkBuffer buffer = VK_NULL_HANDLE; - VkDeviceMemory memory = VK_NULL_HANDLE; - void* mapped = nullptr; - create_host_buffer(vertex_bytes + index_bytes, - VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); - vkUnmapMemory(device_, f_->ui_memory); - vkDestroyBuffer(device_, f_->ui_buffer, nullptr); - vkFreeMemory(device_, f_->ui_memory, nullptr); - f_->ui_buffer = buffer; - f_->ui_memory = memory; - f_->ui_mapped = static_cast(mapped); - f_->ui_vertex_capacity = quads * 4; - f_->ui_index_capacity = quads * 6; - f_->ui_vertex_bytes = vertex_bytes; -} - -void VulkanWindow::create_render_pass_and_layout() { - const bool msaa = samples_ != VK_SAMPLE_COUNT_1_BIT; - VkAttachmentDescription attachments[3]{}; - attachments[0].format = swapchain_format_; attachments[0].samples = samples_; - attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; - attachments[0].storeOp = msaa ? VK_ATTACHMENT_STORE_OP_DONT_CARE : VK_ATTACHMENT_STORE_OP_STORE; - attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; - attachments[0].finalLayout = msaa ? VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL : VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - // MSAA resolve target: the swapchain image. - attachments[2].format = swapchain_format_; attachments[2].samples = VK_SAMPLE_COUNT_1_BIT; - attachments[2].loadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; attachments[2].storeOp = VK_ATTACHMENT_STORE_OP_STORE; - attachments[2].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[2].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; - attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = samples_; - attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; - attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; - attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - - VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; - VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; - VkAttachmentReference resolve_ref{2, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; - VkSubpassDescription subpass{}; - subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; - subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; - if (msaa) subpass.pResolveAttachments = &resolve_ref; - subpass.pDepthStencilAttachment = &depth_ref; - - VkSubpassDependency dependency{}; - dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0; - dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; - dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; - dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; - - VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; - pass_info.attachmentCount = msaa ? 3 : 2; pass_info.pAttachments = attachments; - pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; - pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency; - check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass"); - create_framebuffers(); - - const auto vert = read_spirv("native.vert.spv"); - const auto frag = read_spirv("native.frag.spv"); - auto module = [&](const std::vector& code) { - VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO}; - info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data(); - VkShaderModule handle = VK_NULL_HANDLE; - check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule"); - return handle; - }; - vertex_module_ = module(vert); - fragment_module_ = module(frag); - - VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_}; - VkPushConstantRange push_constants{}; - push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; - push_constants.size = sizeof(PushConstants); - VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO}; - layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts; - layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants; - check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); - - get_pipeline(0, 2, 0); - create_offscreen_pass(); -} - -void VulkanWindow::create_offscreen_pass() { - VkFormatProperties properties{}; - vkGetPhysicalDeviceFormatProperties(physical_, VK_FORMAT_R5G6B5_UNORM_PACK16, &properties); - const VkFormatFeatureFlags wanted = - VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT; - offscreen_format_ = (properties.optimalTilingFeatures & wanted) == wanted - ? VK_FORMAT_R5G6B5_UNORM_PACK16 : VK_FORMAT_R8G8B8A8_UNORM; - VkAttachmentDescription attachments[2]{}; - attachments[0].format = offscreen_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; - attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; - attachments[0].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; - attachments[0].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; - attachments[0].initialLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - attachments[0].finalLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; - attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_STORE; - attachments[1].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; - attachments[1].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; - attachments[1].initialLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; - VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; - VkSubpassDescription subpass{}; - subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; - subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; - subpass.pDepthStencilAttachment = &depth_ref; - VkSubpassDependency dependencies[2]{}; - // Earlier work on the queue -- the previous frame sampling the texture, the previous - // offscreen pass, the first-use clear -- completes before this pass writes it. - dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL; dependencies[0].dstSubpass = 0; - dependencies[0].srcStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | - VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | - VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT; - dependencies[0].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | - VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT; - dependencies[0].dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | - VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT; - dependencies[0].dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | - VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; - // The back-buffer pass samples the result. - dependencies[1].srcSubpass = 0; dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL; - dependencies[1].srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; - dependencies[1].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT; - dependencies[1].dstStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT; - dependencies[1].dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; - pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; - pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; - pass_info.dependencyCount = 2; pass_info.pDependencies = dependencies; - check(vkCreateRenderPass(device_, &pass_info, nullptr, &offscreen_pass_), "vkCreateRenderPass offscreen"); -} - -VulkanWindow::RenderTarget* VulkanWindow::get_render_target(const std::string& name) { - if (auto it = render_targets_.find(name); it != render_targets_.end()) return &it->second; - unsigned id = 0, w = 0, h = 0; - if (std::sscanf(name.c_str(), "rt:%u:%ux%u", &id, &w, &h) != 3 || !w || !h || w > 4096 || h > 4096) - return nullptr; - RenderTarget& target = render_targets_[name]; - target.width = w; - target.height = h; - auto make_image = [&](VkFormat format, VkImageUsageFlags usage, VkImageAspectFlags aspect, - VkImage& image, VkDeviceMemory& memory, VkImageView& view) { - VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - info.imageType = VK_IMAGE_TYPE_2D; - info.format = format; - info.extent = {w, h, 1}; - info.mipLevels = 1; info.arrayLayers = 1; - info.samples = VK_SAMPLE_COUNT_1_BIT; - info.tiling = VK_IMAGE_TILING_OPTIMAL; - info.usage = usage; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &info, nullptr, &image), "vkCreateImage render target"); - VkMemoryRequirements reqs{}; - vkGetImageMemoryRequirements(device_, image, &reqs); - VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - alloc.allocationSize = reqs.size; - alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory render target"); - check(vkBindImageMemory(device_, image, memory, 0), "vkBindImageMemory render target"); - VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view_info.image = image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = format; - view_info.subresourceRange = {aspect, 0, 1, 0, 1}; - check(vkCreateImageView(device_, &view_info, nullptr, &view), "vkCreateImageView render target"); - }; - make_image(offscreen_format_, - VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | - VK_IMAGE_USAGE_TRANSFER_SRC_BIT, - VK_IMAGE_ASPECT_COLOR_BIT, target.texture.image, target.texture.memory, target.texture.view); - make_image(VK_FORMAT_D32_SFLOAT, - VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, - VK_IMAGE_ASPECT_DEPTH_BIT, target.depth_image, target.depth_memory, target.depth_view); - const VkImageView views[2] = {target.texture.view, target.depth_view}; - VkFramebufferCreateInfo fb{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; - fb.renderPass = offscreen_pass_; - fb.attachmentCount = 2; fb.pAttachments = views; - fb.width = w; fb.height = h; fb.layers = 1; - check(vkCreateFramebuffer(device_, &fb, nullptr, &target.framebuffer), "vkCreateFramebuffer render target"); - target.texture.descriptor = allocate_texture_descriptor_set(target.texture.view, fallback_texture_.view); - target.texture.ui_descriptor = allocate_ui_texture_descriptor_set(target.texture.view, fallback_texture_.view); - return ⌖ -} - -void VulkanWindow::release_render_target(RenderTarget& target) { - if (target.framebuffer) vkDestroyFramebuffer(device_, target.framebuffer, nullptr); - if (target.depth_view) vkDestroyImageView(device_, target.depth_view, nullptr); - if (target.depth_image) vkDestroyImage(device_, target.depth_image, nullptr); - if (target.depth_memory) vkFreeMemory(device_, target.depth_memory, nullptr); - release_texture(target.texture); - target = {}; -} - -void VulkanWindow::initialize_render_target(RenderTarget& target) { - VkImageMemoryBarrier barriers[2]{}; - for (auto& b : barriers) { - b.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER; - b.srcQueueFamilyIndex = b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - b.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; - b.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - b.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - } - barriers[0].image = target.texture.image; - barriers[0].subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; - barriers[1].image = target.depth_image; - barriers[1].subresourceRange = {VK_IMAGE_ASPECT_DEPTH_BIT, 0, 1, 0, 1}; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, - 0, 0, nullptr, 0, nullptr, 2, barriers); - const VkClearColorValue white{{1.0f, 1.0f, 1.0f, 1.0f}}; - const VkClearDepthStencilValue far{1.0f, 0}; - vkCmdClearColorImage(f_->command, target.texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &white, 1, - &barriers[0].subresourceRange); - vkCmdClearDepthStencilImage(f_->command, target.depth_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &far, 1, - &barriers[1].subresourceRange); - for (auto& b : barriers) { - b.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - b.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - } - barriers[0].newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - barriers[0].dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_READ_BIT; - barriers[1].newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; - barriers[1].dstAccessMask = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT; - vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT, - VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | - VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, - 0, 0, nullptr, 0, nullptr, 2, barriers); - target.initialized = true; -} - -VkPipeline VulkanWindow::get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, - bool lines, std::uint8_t z_func, bool offscreen) { - const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | - (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | - (std::uint32_t(z_func & 0xfu) << 25) | (offscreen ? (1u << 29) : 0u); - if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; - - VkPipelineShaderStageCreateInfo stages[2]{}; - stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; - stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main"; - stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main"; - - VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX}; - VkVertexInputAttributeDescription attributes[9] = { - {0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)}, - {1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)}, - {2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)}, - {3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)}, - {4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)}, - {5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)}, - {6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)}, - {7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)}, - {8, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, vertex_fog)}}; - VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO}; - input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding; - input.vertexAttributeDescriptionCount = 9; input.pVertexAttributeDescriptions = attributes; - - VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO}; - assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; - - VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; - // An offscreen target is at most the device's 4096 (GetDeviceCaps); the viewport clips inside it. - VkRect2D scissor{{0, 0}, offscreen ? VkExtent2D{4096, 4096} : extent_}; - VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; - viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; - viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; - // D3DVIEWPORT8 changes per draw (SetViewport), so the viewport is dynamic state. - const VkDynamicState dynamic_states[] = {VK_DYNAMIC_STATE_VIEWPORT}; - VkPipelineDynamicStateCreateInfo dynamic_info{VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO}; - dynamic_info.dynamicStateCount = 1; dynamic_info.pDynamicStates = dynamic_states; - - VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO}; - raster.polygonMode = VK_POLYGON_MODE_FILL; - raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE); - raster.frontFace = VK_FRONT_FACE_CLOCKWISE; - raster.lineWidth = 1; - - VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; - multisample.rasterizationSamples = offscreen ? VK_SAMPLE_COUNT_1_BIT : samples_; - - VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; - depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; - depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE; - switch (z_func) { - case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break; - case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break; - case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break; - case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break; - case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break; - case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break; - case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break; - default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break; - } - - auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor { - switch (d3d) { - case 1: return VK_BLEND_FACTOR_ZERO; - case 2: return VK_BLEND_FACTOR_ONE; - case 3: return VK_BLEND_FACTOR_SRC_COLOR; - case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR; - case 5: return VK_BLEND_FACTOR_SRC_ALPHA; - case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; - case 7: return VK_BLEND_FACTOR_DST_ALPHA; - case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; - case 9: return VK_BLEND_FACTOR_DST_COLOR; - case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR; - case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE; - default: return fallback; - } - }; - - VkPipelineColorBlendAttachmentState blend_attachment{}; - blend_attachment.colorWriteMask = 0xf; - if (blend > 0) { - blend_attachment.blendEnable = VK_TRUE; - if (blend >= 16) { - blend_attachment.srcColorBlendFactor = - d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA); - blend_attachment.dstColorBlendFactor = - d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA); - } else { - blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA; - blend_attachment.dstColorBlendFactor = - blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; - } - blend_attachment.colorBlendOp = VK_BLEND_OP_ADD; - // D3D8 has no separate alpha blend: SRCBLEND/DESTBLEND apply to alpha as well. - auto alpha_factor = [](VkBlendFactor factor) { - switch (factor) { - case VK_BLEND_FACTOR_SRC_COLOR: return VK_BLEND_FACTOR_SRC_ALPHA; - case VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; - case VK_BLEND_FACTOR_DST_COLOR: return VK_BLEND_FACTOR_DST_ALPHA; - case VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; - case VK_BLEND_FACTOR_SRC_ALPHA_SATURATE: return VK_BLEND_FACTOR_ONE; - default: return factor; - } - }; - blend_attachment.srcAlphaBlendFactor = alpha_factor(blend_attachment.srcColorBlendFactor); - blend_attachment.dstAlphaBlendFactor = alpha_factor(blend_attachment.dstColorBlendFactor); - blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD; - } - VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO}; - blending.attachmentCount = 1; blending.pAttachments = &blend_attachment; - - VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO}; - pipeline_info.stageCount = 2; pipeline_info.pStages = stages; - pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly; - pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster; - pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; - pipeline_info.pColorBlendState = &blending; - pipeline_info.pDynamicState = &dynamic_info; - pipeline_info.layout = layout_; pipeline_info.renderPass = offscreen ? offscreen_pass_ : pass_; - VkPipeline handle = VK_NULL_HANDLE; - check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); - pipelines_.emplace(key, handle); - return handle; -} - -VkDescriptorSet VulkanWindow::allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, - VkSampler sampler0, - VkSampler sampler1) { - VkDescriptorSet set = VK_NULL_HANDLE; - VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - desc_alloc.descriptorPool = descriptor_pool_; - desc_alloc.descriptorSetCount = 1; - desc_alloc.pSetLayouts = &descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets"); - - VkDescriptorImageInfo image_infos[2]{}; - image_infos[0].sampler = sampler0 ? sampler0 : sampler_; - image_infos[0].imageView = view0; - image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_); - image_infos[1].imageView = view1; - image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - - VkWriteDescriptorSet writes[2]{}; - for (int b = 0; b < 2; ++b) { - writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[b].dstSet = set; - writes[b].dstBinding = static_cast(b); - writes[b].descriptorCount = 1; - writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - writes[b].pImageInfo = &image_infos[b]; - } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); - return set; -} - -VkDescriptorSet VulkanWindow::allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) { - VkDescriptorSet set = VK_NULL_HANDLE; - VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; - desc_alloc.descriptorPool = descriptor_pool_; - desc_alloc.descriptorSetCount = 1; - desc_alloc.pSetLayouts = &descriptor_layout_; - check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui"); - - VkDescriptorImageInfo image_infos[2]{}; - image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_; - image_infos[0].imageView = view0; - image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_); - image_infos[1].imageView = view1; - image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - - VkWriteDescriptorSet writes[2]{}; - for (int b = 0; b < 2; ++b) { - writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; - writes[b].dstSet = set; - writes[b].dstBinding = static_cast(b); - writes[b].descriptorCount = 1; - writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; - writes[b].pImageInfo = &image_infos[b]; - } - vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); - return set; -} - -VulkanWindow::GpuTexture& VulkanWindow::get_gpu_texture( - const std::string& texture_name, - const std::unordered_map>& capture_textures) { - if (texture_name.empty()) return fallback_texture_; - if (texture_name.rfind("rt:", 0) == 0) { - RenderTarget* target = get_render_target(texture_name); - return target ? target->texture : fallback_texture_; - } - auto [it, inserted] = textures_.try_emplace(texture_name); - it->second.last_used_frame = frame_number_; - if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) - return it->second; - if (inserted) { - it->second.descriptor = fallback_texture_.descriptor; - it->second.ui_descriptor = fallback_texture_.ui_descriptor; - } - it->second.last_decode_attempt_frame = frame_number_; - - mtimage::Image decoded; - if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) { - decoded = decode_texture_bytes(raw->second.data(), raw->second.size()); - } -#ifdef MT_NATIVE_HAS_LIVE_CLIENT - if (!decoded.ok()) { - if (texture_name.rfind("mem:", 0) == 0) { - UIMemoryTexture mem_tex; - if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { - const auto mtra = native_draw_capture::encode_raw_argb_as_mtra( - static_cast(mem_tex.width), - static_cast(mem_tex.height), - mem_tex.argb.data()); - decoded = decode_texture_bytes(mtra.data(), mtra.size()); - } - } else { - std::vector pack_bytes; - if (read_live_pack_texture(texture_name, pack_bytes)) - decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size()); - } - } -#endif - if (!decoded.ok()) return it->second; - // 40250 CGraphicImageTexture: DXT DDS -> CreateDDSTexture with the file's levels; - // other files -> D3DXCreateTextureFromFileInMemoryEx(D3DX_DEFAULT) = full chain; - // memory images (mem:, CreateTexture(w, h, 1, ...)) -> one level. - const bool is_memory_tex = texture_name.rfind("mem:", 0) == 0; - last_texture_upload_name_ = texture_name; - upload_texture( - decoded.w, decoded.h, decoded.rgba.data(), it->second, true, - !is_memory_tex && decoded.file_mip_levels == 0, &decoded.mips); - it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view); - it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view); - return it->second; -} - -std::uint64_t VulkanWindow::sampler_state_key(const Render3DDraw& draw, int stage) const { - return (std::uint64_t(draw.address_u[stage] & 15u)) | - (std::uint64_t(draw.address_v[stage] & 15u) << 4) | - (std::uint64_t(draw.min_filter[stage] & 15u) << 8) | - (std::uint64_t(draw.mag_filter[stage] & 15u) << 12) | - (std::uint64_t(draw.mip_filter[stage] & 15u) << 16); -} - -VkSampler VulkanWindow::sampler_for_draw(const Render3DDraw& draw, int stage) { - const auto key = sampler_state_key(draw, stage); - if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_; - if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second; - auto address_mode = [](std::uint32_t mode) { - switch (mode) { - case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT; - case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; - case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER; - default: return VK_SAMPLER_ADDRESS_MODE_REPEAT; - } - }; - VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; - info.addressModeU = address_mode(draw.address_u[stage]); - info.addressModeV = address_mode(draw.address_v[stage]); - info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; - info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE; - info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; - info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; - info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR; - info.minLod = 0.0f; - info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE; - if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) { - info.anisotropyEnable = VK_TRUE; - info.maxAnisotropy = max_anisotropy_; - } - VkSampler sampler = VK_NULL_HANDLE; - check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state"); - state_samplers_.emplace(key, sampler); - return sampler; -} - -VkDescriptorSet VulkanWindow::get_texture_descriptor( - const std::string& texture0_name, - const std::string& mask_name, - const std::unordered_map>& capture_textures, - const Render3DDraw& draw) { - const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); - const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures); - const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" + - std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1)); - if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { - it->second.last_used_frame = frame_number_; - return it->second.set; - } - const VkDescriptorSet set = allocate_texture_descriptor_set( - tex0.view ? tex0.view : fallback_texture_.view, - tex1.view ? tex1.view : fallback_texture_.view, - sampler_for_draw(draw, 0), sampler_for_draw(draw, 1)); - paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); - return set; -} - -VkDescriptorSet VulkanWindow::get_ui_texture_descriptor( - const std::string& texture0_name, - const std::string& mask_name, - const std::unordered_map>& capture_textures) { - const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); - if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; - const auto& tex1 = get_gpu_texture(mask_name, capture_textures); - if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; - const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name; - if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { - it->second.last_used_frame = frame_number_; - return it->second.set; - } - const VkDescriptorSet set = allocate_ui_texture_descriptor_set( - tex0.view ? tex0.view : fallback_texture_.view, - tex1.view ? tex1.view : fallback_texture_.view); - paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); - return set; -} - -void VulkanWindow::upload_texture( - std::uint32_t width, - std::uint32_t height, - const std::uint8_t* rgba, - GpuTexture& texture, - bool count_upload, - bool generate_mips, - const std::vector>* file_mips) { - const ScopedMs timer(timings_.texture_upload_ms); - const bool has_file_mips = !generate_mips && file_mips && !file_mips->empty(); - VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; - if (has_file_mips) - for (const auto& level : *file_mips) byte_size += level.size(); - const std::uint32_t mip_levels = generate_mips - ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) - : has_file_mips ? 1u + static_cast(file_mips->size()) : 1u; - // Inside a frame the copy is recorded ahead of that frame's render pass and reads the slot's - // staging ring; the GPU runs it with the frame, no queue wait. Outside a frame (the fallback - // texture at startup) it is a one-off submit that waits. - const bool in_frame = f_->recording; - VkBuffer staging_buffer = VK_NULL_HANDLE; - VkDeviceMemory staging_memory = VK_NULL_HANDLE; - VkDeviceSize staging_offset = 0; - std::uint8_t* mapped = nullptr; - const VkDeviceSize aligned_offset = - (f_->staging_used + staging_alignment_ - 1) / staging_alignment_ * staging_alignment_; - if (in_frame && f_->staging_mapped && aligned_offset + byte_size <= f_->staging_capacity) { - staging_buffer = f_->staging_buffer; - staging_offset = aligned_offset; - mapped = f_->staging_mapped + staging_offset; - f_->staging_used = aligned_offset + byte_size; - } else { - if (in_frame) f_->staging_wanted = std::max(f_->staging_wanted, aligned_offset + byte_size); - void* raw = nullptr; - create_host_buffer(byte_size, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, staging_buffer, staging_memory, &raw); - mapped = static_cast(raw); - } - std::memcpy(mapped, rgba, VkDeviceSize(width) * height * 4); - if (has_file_mips) { - auto* dst = mapped + VkDeviceSize(width) * height * 4; - for (const auto& level : *file_mips) { std::memcpy(dst, level.data(), level.size()); dst += level.size(); } - } - if (staging_memory) vkUnmapMemory(device_, staging_memory); - - VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; - image_info.imageType = VK_IMAGE_TYPE_2D; - image_info.format = VK_FORMAT_R8G8B8A8_UNORM; - image_info.extent = {width, height, 1}; - image_info.mipLevels = mip_levels; - image_info.arrayLayers = 1; - image_info.samples = VK_SAMPLE_COUNT_1_BIT; - image_info.tiling = VK_IMAGE_TILING_OPTIMAL; - image_info.usage = - VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT; - image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture"); - VkMemoryRequirements image_reqs{}; - vkGetImageMemoryRequirements(device_, texture.image, &image_reqs); - VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - image_alloc.allocationSize = image_reqs.size; - image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); - check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); - check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); - - VkCommandBuffer upload_cmd = f_->command; - if (!in_frame) { - VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; - cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; - check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); - VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; - begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; - check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); - } - - VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; - to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_transfer.image = texture.image; - to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &to_transfer); - - std::vector regions(has_file_mips ? mip_levels : 1u); - VkDeviceSize region_offset = 0; - for (std::uint32_t i = 0; i < regions.size(); ++i) { - const std::uint32_t lw = std::max(1u, width >> i), lh = std::max(1u, height >> i); - regions[i].bufferOffset = staging_offset + region_offset; - regions[i].imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; - regions[i].imageExtent = {lw, lh, 1}; - region_offset += VkDeviceSize(lw) * lh * 4; - } - vkCmdCopyBufferToImage( - upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, - static_cast(regions.size()), regions.data()); - - std::int32_t mip_w = static_cast(width); - std::int32_t mip_h = static_cast(height); - for (std::uint32_t i = 1; generate_mips && i < mip_levels; ++i) { - VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - barrier.image = texture.image; - barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1}; - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &barrier); - - const std::int32_t next_w = std::max(1, mip_w / 2); - const std::int32_t next_h = std::max(1, mip_h / 2); - VkImageBlit blit{}; - blit.srcOffsets[0] = {0, 0, 0}; - blit.srcOffsets[1] = {mip_w, mip_h, 1}; - blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1}; - blit.dstOffsets[0] = {0, 0, 0}; - blit.dstOffsets[1] = {next_w, next_h, 1}; - blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; - vkCmdBlitImage( - upload_cmd, - texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, - texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, - 1, &blit, VK_FILTER_LINEAR); - - barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; - barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &barrier); - - mip_w = next_w; - mip_h = next_h; - } - - VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; - to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; - to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; - to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; - to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; - to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; - to_shader.image = texture.image; - to_shader.subresourceRange = generate_mips - ? VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1} - : VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; - vkCmdPipelineBarrier( - upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, - 0, 0, nullptr, 0, nullptr, 1, &to_shader); - - if (in_frame) { - if (staging_memory) f_->retired_buffers.emplace_back(staging_buffer, staging_memory); - } else { - check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); - VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; - submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; - check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); - check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); - vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); - vkDestroyBuffer(device_, staging_buffer, nullptr); - vkFreeMemory(device_, staging_memory, nullptr); - } - - VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; - view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; - view_info.format = VK_FORMAT_R8G8B8A8_UNORM; - view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; - check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture"); - - if (count_upload) { - ++texture_upload_count_; - texture_uploaded_bytes_ += byte_size; - } -} - -void VulkanWindow::release_texture(GpuTexture& texture) { - for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor}) - if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor) - vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set); - if (texture.view) vkDestroyImageView(device_, texture.view, nullptr); - if (texture.image) vkDestroyImage(device_, texture.image, nullptr); - if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr); - texture = {}; -} - -void VulkanWindow::prune_textures() { - constexpr std::uint64_t static_retention_frames = 120; - constexpr std::uint64_t memory_retention_frames = 2; - for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) { - const auto retention = it->first.find("mem:") != std::string::npos - ? memory_retention_frames : static_retention_frames; - if (frame_number_ - it->second.last_used_frame > retention) { - vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set); - it = paired_descriptors_.erase(it); - } else ++it; - } - for (auto it = textures_.begin(); it != textures_.end();) { - const auto retention = it->first.rfind("mem:", 0) == 0 - ? memory_retention_frames : static_retention_frames; - if (frame_number_ - it->second.last_used_frame > retention) { - release_texture(it->second); - it = textures_.erase(it); - } else ++it; - } -} - -void VulkanWindow::release_geometry(Geometry& geometry) { - if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr); - if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr); - geometry = {}; -} - -void VulkanWindow::upload_geometry(const Render3DDraw& draw, Geometry& geometry) { - const ScopedMs timer(timings_.geometry_upload_ms); - if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) - throw std::runtime_error("invalid native geometry array size"); - const auto count = draw.positions.size() / 3; - if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX) - throw std::runtime_error("geometry exceeds Vulkan index range"); - // D3D8 feeds opaque white for a vertex format without D3DFVF_DIFFUSE. - const std::uint32_t default_argb = 0xffffffffu; - const bool has_skin = - draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4; - std::vector vertices(count); - for (std::size_t i = 0; i < count; ++i) { - vertices[i].position[0] = draw.positions[3*i]; - vertices[i].position[1] = draw.positions[3*i+1]; - vertices[i].position[2] = draw.positions[3*i+2]; - vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f; - vertices[i].vertex_fog = i < draw.vertex_fog.size() ? draw.vertex_fog[i] : 1.0f; - if (3*i + 2 < draw.normals.size()) { - vertices[i].normal[0] = draw.normals[3*i]; - vertices[i].normal[1] = draw.normals[3*i+1]; - vertices[i].normal[2] = draw.normals[3*i+2]; - } else { - vertices[i].normal[0] = 0.0f; - vertices[i].normal[1] = 0.0f; - vertices[i].normal[2] = 1.0f; - } - if (2*i + 1 < draw.uv0.size()) { - vertices[i].uv[0] = draw.uv0[2*i]; - vertices[i].uv[1] = draw.uv0[2*i+1]; - } else { - vertices[i].uv[0] = 0.0f; - vertices[i].uv[1] = 0.0f; - } - const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb; - vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f; - vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f; - vertices[i].color[2] = float(argb & 255) / 255.0f; - vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f; - if (has_skin) { - for (int k = 0; k < 4; ++k) { - vertices[i].joints[k] = draw.bone_indices[4*i + k]; - vertices[i].weights[k] = draw.bone_weights[4*i + k]; - } - } else { - vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0; - vertices[i].weights[0] = 1.0f; - vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f; - } - if (2*i + 1 < draw.uv1.size()) { - vertices[i].mask_uv[0] = draw.uv1[2*i]; - vertices[i].mask_uv[1] = draw.uv1[2*i+1]; - } else { - vertices[i].mask_uv[0] = 0.0f; - vertices[i].mask_uv[1] = 0.0f; - } - } - for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count"); - geometry.index_offset = vertices.size() * sizeof(Vertex); - const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t); - if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB"); - VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; - info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT; - info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer"); - VkMemoryRequirements requirements{}; - vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements); - VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; - allocation.allocationSize = requirements.size; - allocation.memoryTypeIndex = find_memory_type( - requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); - check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory"); - check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory"); - void* mapped = nullptr; - check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory"); - std::memcpy(mapped, vertices.data(), geometry.index_offset); - std::memcpy(static_cast(mapped) + geometry.index_offset, - draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); - vkUnmapMemory(device_, geometry.memory); - geometry.vertex_count = static_cast(count); - geometry.index_count = static_cast(draw.indices.size()); - ++upload_count_; - uploaded_bytes_ += bytes; -} - -void VulkanWindow::update_fps_counter() { - const auto now = std::chrono::steady_clock::now(); - if (fps_frames_++ == 0) { - fps_window_start_ = now; - return; - } - const double elapsed = std::chrono::duration(now - fps_window_start_).count(); - if (elapsed >= 0.5) { - fps_ = int(std::lround((fps_frames_ - 1) / elapsed)); - fps_frames_ = 1; - fps_window_start_ = now; - } -} - -void VulkanWindow::note_input(const SDL_Event& event) { - switch (event.type) { - case SDL_EVENT_FINGER_DOWN: ++fingers_down_; break; - case SDL_EVENT_FINGER_UP: - case SDL_EVENT_FINGER_CANCELED: fingers_down_ = std::max(0, fingers_down_ - 1); break; - case SDL_EVENT_MOUSE_BUTTON_DOWN: ++buttons_down_; break; - case SDL_EVENT_MOUSE_BUTTON_UP: buttons_down_ = std::max(0, buttons_down_ - 1); break; - case SDL_EVENT_WINDOW_FOCUS_LOST: case SDL_EVENT_WILL_ENTER_BACKGROUND: - // An up event lost with the focus must not keep the client out of the idle rate for good. - fingers_down_ = buttons_down_ = 0; - break; - case SDL_EVENT_FINGER_MOTION: case SDL_EVENT_MOUSE_MOTION: case SDL_EVENT_MOUSE_WHEEL: - case SDL_EVENT_KEY_DOWN: case SDL_EVENT_KEY_UP: case SDL_EVENT_TEXT_INPUT: - case SDL_EVENT_WINDOW_FOCUS_GAINED: case SDL_EVENT_DID_ENTER_FOREGROUND: break; - default: return; - } - last_input_ = std::chrono::steady_clock::now(); -} - void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms, std::vector frame_ms) { const auto timing = renderer.timings(); diff --git a/src/host/vulkan_window.h b/src/host/vulkan_window.h index 138353cc..1fa0170c 100644 --- a/src/host/vulkan_window.h +++ b/src/host/vulkan_window.h @@ -196,6 +196,44 @@ public: native_perf::RendererTotals perf_totals() const; private: + using TextureBytes = std::unordered_map>; + + // render()'s phases, in call order (vk_frame.cpp). + struct FrameDraws { + std::vector prepared_draws; // the back-buffer pass + // Draws into offscreen render targets (the shadow map), in recorded order, drawn before the + // back-buffer pass that samples them. + std::vector offscreen_draws; + std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; + }; + // Host copies of the swapchain image (--screenshot-out) and, with MT_DUMP_RT, of every render target. + struct Readback { + struct RtDump { VkBuffer buffer; VkDeviceMemory memory; std::uint32_t w, h; }; + bool screenshot = false; + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + const char* dump_rt = nullptr; + std::vector rt_dumps; + }; + + // Geometry, pipeline and descriptors of each recorded draw; retires unused geometry. + void prepare_draws(const std::vector& draws, const TextureBytes& capture_textures, FrameDraws& frame); + + // The UI batches, with the touch controls and FPS counter appended to the 40250 UI. + void prepare_ui(std::uint32_t ui_width, std::uint32_t ui_height, const std::vector& ui_commands, + const TextureBytes& capture_textures, std::vector& ui_batches, std::size_t& ui_quad_count); + + // Each draw's FixedFunctionState into the frame's dynamic-offset state buffer. + void write_draw_states(FrameDraws& frame, std::vector& ui_batches); + + // Offscreen passes, then the back-buffer pass: UI behind 3D, 3D, UI. + void record_passes(std::uint32_t image_index, const FrameDraws& frame, const std::vector& ui_batches); + + // Recorded after the passes; finish_readbacks() waits for the submit and writes the files. + Readback record_readbacks(std::uint32_t image_index); + + void finish_readbacks(Readback& readback); + void recreate_swapchain(); void collect_pending_gpu_timestamp(FrameSlot& slot);