#include "RenderCommands3D.h" #include "UIRenderCommands.h" #include "draw_capture.h" #include "dxt.h" #include "touch_controller.h" #ifdef MT_NATIVE_HAS_LIVE_CLIENT #include "platform/MilesLib/AudioCommands.h" #include "platform/PackBackend.h" #include "platform/ScriptLib/PythonBoot.h" #include "platform/UserInterface/ServerClock.h" #include "../tests/port_login_flow_server.h" #include #endif #include #include #include #include #include #include "stb_image.h" #ifdef __APPLE__ #include #endif #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include namespace { void check(VkResult result, const char* operation) { if (result != VK_SUCCESS) throw std::runtime_error(std::string(operation) + ": VkResult " + std::to_string(result)); } struct Vertex { float position[3]; float normal[3]; float uv[2]; float color[4]; std::uint8_t joints[4]; float weights[4]; float mask_uv[2]; float rhw = 1.0f; }; struct PushConstants { std::array mvp{}; std::array tint_color{1.0f, 1.0f, 1.0f, 1.0f}; std::array ambient_emissive{1.0f, 1.0f, 1.0f, -1.0f}; std::array light_dir{0.0f, 0.0f, 1.0f, 0.0f}; std::array light_diffuse{0.0f, 0.0f, 0.0f, 0.0f}; }; static_assert(sizeof(PushConstants) == 128, "PushConstants must fit within Vulkan's 128-byte minimum guarantee"); // std430 payload selected with a dynamic storage-buffer offset for every draw. // The existing 128-byte push constants cannot hold the D3D8 texture stages and fog state. struct alignas(16) FixedFunctionState { std::array stage0_color{}; // op, arg1, arg2, unused std::array stage0_alpha{}; // op, arg1, arg2, unused std::array stage1_color{}; std::array stage1_alpha{}; std::array fog_color{}; std::array fog_params{}; // start, end, density, range enabled std::array texture_factor{}; std::array world_view{}; std::array flags{}; // fixed function, texture0, texture1, fog mode std::array world{}; std::array lighting_flags{}; // enabled, COLORVERTEX, ambient source, diffuse source std::array material_ambient{}; std::array material_emissive{}; std::array global_ambient{}; std::array, 2> light_position_type{}; std::array, 2> light_direction_range{}; std::array, 2> light_diffuse{}; std::array, 2> light_ambient{}; std::array, 2> light_attenuation{}; std::array, 2> light_spot{}; }; static_assert(sizeof(FixedFunctionState) == 512); constexpr std::array kIdentityMatrix = { 1.0f, 0.0f, 0.0f, 0.0f, 0.0f, 1.0f, 0.0f, 0.0f, 0.0f, 0.0f, 1.0f, 0.0f, 0.0f, 0.0f, 0.0f, 1.0f}; std::array multiply(const float* a, const float* b) { std::array result{}; for (int row = 0; row < 4; ++row) for (int col = 0; col < 4; ++col) for (int k = 0; k < 4; ++k) result[row * 4 + col] += a[row * 4 + k] * b[k * 4 + col]; return result; } std::array draw_mvp(const Render3DDraw& draw) { const auto world_view = multiply(draw.world, draw.view); return multiply(world_view.data(), draw.proj); } bool draw_is_specular_spheremap(const Render3DDraw& draw) { // D3DTOP_MODULATEALPHA_ADDCOLOR == 18 on texture stage 1 return !draw.texture1.empty() && draw.color_op[1] == 18; } std::array unpack_argb(std::uint32_t argb) { return { float((argb >> 16) & 255) / 255.0f, float((argb >> 8) & 255) / 255.0f, float(argb & 255) / 255.0f, float((argb >> 24) & 255) / 255.0f}; } PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded) { PushConstants constants{}; constants.mvp = draw_mvp(draw); const bool lit = draw.lighting != 0; for (int c = 0; c < 4; ++c) { const float base = lit ? draw.material_diffuse[c] : 1.0f; constants.tint_color[c] = base; } const float alpha_cutoff = draw.alpha_test ? (float(draw.alpha_ref) / 255.0f) : -1.0f; if (lit && draw.light0) { const auto amb = unpack_argb(draw.ambient); for (int c = 0; c < 3; ++c) constants.ambient_emissive[c] = draw.material_ambient[c] * std::min(amb[c] + draw.light0_ambient[c], 1.0f) + draw.material_emissive[c]; constants.ambient_emissive[3] = alpha_cutoff; const float lx = -draw.light0_direction[0]; const float ly = -draw.light0_direction[1]; const float lz = -draw.light0_direction[2]; float mx = draw.world[0] * lx + draw.world[1] * ly + draw.world[2] * lz; float my = draw.world[4] * lx + draw.world[5] * ly + draw.world[6] * lz; float mz = draw.world[8] * lx + draw.world[9] * ly + draw.world[10] * lz; const float len = std::sqrt(mx * mx + my * my + mz * mz); if (len > 1e-6f) { mx /= len; my /= len; mz /= len; } else { mx = 0.0f; my = 0.0f; mz = 1.0f; } constants.light_dir = {mx, my, mz, 1.0f}; constants.light_diffuse = { draw.light0_diffuse[0], draw.light0_diffuse[1], draw.light0_diffuse[2], skin_offset_encoded}; } else if (lit) { const auto amb = unpack_argb(draw.ambient); for (int c = 0; c < 3; ++c) constants.ambient_emissive[c] = draw.material_ambient[c] * amb[c] + draw.material_emissive[c]; constants.ambient_emissive[3] = alpha_cutoff; constants.light_dir = {0.0f, 0.0f, 1.0f, 1.0f}; constants.light_diffuse[3] = skin_offset_encoded; } else { constants.ambient_emissive = {1.0f, 1.0f, 1.0f, alpha_cutoff}; constants.light_diffuse[3] = skin_offset_encoded; } if (draw_is_specular_spheremap(draw)) { constants.light_dir[3] = 2.0f; } else if (!draw.texture1.empty() && (draw.color_op[1] > 1 || draw.alpha_op[1] > 1)) { constants.light_dir[3] = lit ? -0.6f : -1.0f; } if (draw.pretransformed) constants.light_diffuse[3] = -1.0f; return constants; } FixedFunctionState make_fixed_function_state(const Render3DDraw& draw) { FixedFunctionState state{}; for (int stage = 0; stage < 2; ++stage) { auto& color = stage ? state.stage1_color : state.stage0_color; auto& alpha = stage ? state.stage1_alpha : state.stage0_alpha; color = {draw.color_op[stage], draw.color_arg1[stage], draw.color_arg2[stage], 0}; alpha = {draw.alpha_op[stage], draw.alpha_arg1[stage], draw.alpha_arg2[stage], stage == 0 ? draw.alpha_func : draw.alpha_test}; } state.texture_factor = unpack_argb(draw.texture_factor); state.world_view = multiply(draw.world, draw.view); const auto fog = unpack_argb(draw.fog_color); state.fog_color = fog; state.fog_params = {draw.fog_start, draw.fog_end, draw.fog_density, draw.fog_range_enable ? 1.0f : 0.0f}; const std::uint32_t fog_mode = draw.fog_enable ? (draw.fog_table_mode ? draw.fog_table_mode : draw.fog_vertex_mode) : 0; const bool multiplicative_shadow = draw.alpha_blend && draw.src_blend == 1 && draw.dest_blend == 3 && !draw.texture0.empty() && draw.color_op[0] == 4; state.flags = {multiplicative_shadow ? 2u : 1u, !draw.texture0.empty() ? 1u : 0u, !draw.texture1.empty() ? 1u : 0u, fog_mode}; std::copy(draw.world, draw.world + 16, state.world.begin()); state.lighting_flags = {draw.lighting, draw.color_vertex, draw.ambient_material_source, draw.diffuse_material_source}; std::copy(draw.material_ambient, draw.material_ambient + 4, state.material_ambient.begin()); std::copy(draw.material_emissive, draw.material_emissive + 4, state.material_emissive.begin()); state.global_ambient = unpack_argb(draw.ambient); for (int i = 0; i < 2; ++i) { const auto& light = draw.lights[i]; state.light_position_type[i] = {light.position[0], light.position[1], light.position[2], float(light.type)}; state.light_direction_range[i] = {light.direction[0], light.direction[1], light.direction[2], light.range}; std::copy(light.diffuse, light.diffuse + 4, state.light_diffuse[i].begin()); std::copy(light.ambient, light.ambient + 4, state.light_ambient[i].begin()); state.light_attenuation[i] = {light.attenuation[0], light.attenuation[1], light.attenuation[2], light.falloff}; state.light_spot[i] = {light.theta, light.phi, 0, 0}; } return state; } std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size) { const auto* data = static_cast(bytes); std::size_t i = 0; for (; i + sizeof(std::uint64_t) <= size; i += sizeof(std::uint64_t)) { std::uint64_t word = 0; std::memcpy(&word, data + i, sizeof(word)); hash = (hash ^ word) * 1099511628211ull; } for (; i < size; ++i) hash = (hash ^ data[i]) * 1099511628211ull; hash = (hash ^ size) * 1099511628211ull; return hash; } // Immediate-mode draws have no source-buffer key. For them, verify content before reusing a slot. std::uint64_t geometry_hash(const Render3DDraw& draw) { std::uint64_t hash = 14695981039346656037ull; const std::uint32_t flags = (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u); hash = hash_bytes(hash, &flags, sizeof(flags)); hash = hash_bytes(hash, draw.positions.data(), draw.positions.size() * sizeof(float)); hash = hash_bytes(hash, draw.rhw.data(), draw.rhw.size() * sizeof(float)); hash = hash_bytes(hash, draw.normals.data(), draw.normals.size() * sizeof(float)); hash = hash_bytes(hash, draw.uv0.data(), draw.uv0.size() * sizeof(float)); hash = hash_bytes(hash, draw.uv1.data(), draw.uv1.size() * sizeof(float)); hash = hash_bytes(hash, draw.diffuse.data(), draw.diffuse.size() * sizeof(std::uint32_t)); return hash_bytes(hash, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); } mtgodot::Image decode_tga(const std::uint8_t* data, std::size_t size) { mtgodot::Image out; if (size < 18) return out; const std::uint8_t id_len = data[0]; const std::uint8_t cmap_type = data[1]; const std::uint8_t img_type = data[2]; const std::uint32_t w = std::uint32_t(data[12]) | (std::uint32_t(data[13]) << 8); const std::uint32_t h = std::uint32_t(data[14]) | (std::uint32_t(data[15]) << 8); const std::uint8_t bpp = data[16]; const std::uint8_t desc = data[17]; if (cmap_type != 0 || !w || !h || w > 4096 || h > 4096) return out; if (img_type != 2 && img_type != 3 && img_type != 10) return out; const std::size_t bytes_per_pixel = bpp / 8; if (bytes_per_pixel != 1 && bytes_per_pixel != 3 && bytes_per_pixel != 4) return out; std::size_t offset = 18 + std::size_t(id_len); if (offset > size) return out; const std::size_t pixel_count = std::size_t(w) * std::size_t(h); std::vector temp(pixel_count * 4); auto write_pixel = [&](std::size_t idx, const std::uint8_t* src) { std::uint8_t* dst = &temp[idx * 4]; if (bytes_per_pixel == 1) { dst[0] = dst[1] = dst[2] = src[0]; dst[3] = 255; } else if (bytes_per_pixel == 3) { dst[0] = src[2]; dst[1] = src[1]; dst[2] = src[0]; dst[3] = 255; } else { dst[0] = src[2]; dst[1] = src[1]; dst[2] = src[0]; dst[3] = src[3]; } }; if (img_type == 2 || img_type == 3) { if (offset + pixel_count * bytes_per_pixel > size) return out; for (std::size_t i = 0; i < pixel_count; ++i) write_pixel(i, data + offset + i * bytes_per_pixel); } else if (img_type == 10) { std::size_t i = 0; while (i < pixel_count && offset < size) { const std::uint8_t header = data[offset++]; const std::size_t run = (header & 0x7fu) + 1u; if (header & 0x80u) { if (offset + bytes_per_pixel > size) return out; for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) write_pixel(i, data + offset); offset += bytes_per_pixel; } else { if (offset + run * bytes_per_pixel > size) return out; for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) { write_pixel(i, data + offset); offset += bytes_per_pixel; } } } if (i != pixel_count) return out; } out.w = static_cast(w); out.h = static_cast(h); const bool top_origin = (desc & 0x20u) != 0; if (top_origin) { out.rgba = std::move(temp); } else { out.rgba.resize(pixel_count * 4); const std::size_t row_bytes = std::size_t(w) * 4; for (std::uint32_t y = 0; y < h; ++y) std::memcpy(&out.rgba[std::size_t(y) * row_bytes], &temp[std::size_t(h - 1 - y) * row_bytes], row_bytes); } return out; } mtgodot::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size) { if (!data || size < 12) return {}; if (data[0] == 'M' && data[1] == 'T' && data[2] == 'R' && data[3] == 'A') { std::uint32_t w = 0, h = 0; std::memcpy(&w, data + 4, 4); std::memcpy(&h, data + 8, 4); const std::size_t bytes = std::size_t(w) * std::size_t(h) * 4; if (w > 0 && h > 0 && w <= 4096 && h <= 4096 && size == 12 + bytes) { mtgodot::Image out; out.w = static_cast(w); out.h = static_cast(h); out.rgba.assign(data + 12, data + 12 + bytes); return out; } return {}; } if (data[0] == 'D' && data[1] == 'D' && data[2] == 'S' && data[3] == ' ') return mtgodot::load_dds(data, size); auto tga = decode_tga(data, size); if (tga.ok()) return tga; if (size > static_cast(INT_MAX)) return {}; int w = 0, h = 0, channels = 0; stbi_uc* pixels = stbi_load_from_memory(data, static_cast(size), &w, &h, &channels, 4); if (pixels && w > 0 && h > 0 && w <= 4096 && h <= 4096) { mtgodot::Image out; out.w = static_cast(w); out.h = static_cast(h); out.rgba.assign(pixels, pixels + std::size_t(w) * std::size_t(h) * 4); stbi_image_free(pixels); return out; } stbi_image_free(pixels); return {}; } #ifdef MT_NATIVE_HAS_LIVE_CLIENT bool read_live_pack_texture(const std::string& vpath, std::vector& bytes) { if (vpath.empty() || !mtpack40250::ready()) return false; std::string norm = vpath; for (char& ch : norm) if (ch == '\\') ch = '/'; std::string stripped = norm; if (stripped.size() >= 2 && stripped[1] == ':') stripped = stripped.substr(2); while (!stripped.empty() && stripped.front() == '/') stripped.erase(stripped.begin()); auto lower = [](std::string s) { for (char& ch : s) if (ch >= 'A' && ch <= 'Z') ch = static_cast(ch - 'A' + 'a'); return s; }; for (const std::string& candidate : { norm, stripped, "d:/" + stripped, lower(norm), lower(stripped), lower("d:/" + stripped)}) { if (mtpack40250::read(candidate, bytes) && !bytes.empty()) return true; } return false; } SDL_Cursor* load_game_cursor(int shape) { static constexpr std::array images = { "cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub", "cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub", "cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub", "cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"}; if (shape < 0 || shape >= static_cast(images.size())) return nullptr; const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape]; std::vector sub_bytes; if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr; std::unordered_map tokens; std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end())); std::string line; while (std::getline(input, line)) { std::istringstream fields(line); std::string key, value; if (!(fields >> key >> value)) continue; if (!value.empty() && value.front() == '"') { value.erase(0, 1); if (!value.empty() && value.back() == '"') value.pop_back(); } std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); }); tokens[key] = value; } if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr; const std::string image_path = tokens["version"] == "2.0" ? std::string("d:/ymir work/ui/cursor/") + tokens["image"] : std::string("d:/ymir work/ui/") + tokens["image"]; std::vector image_bytes; if (!read_live_pack_texture(image_path, image_bytes)) return nullptr; const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size()); if (!image.ok()) return nullptr; const int left = std::atoi(tokens["left"].c_str()); const int top = std::atoi(tokens["top"].c_str()); const int right = std::atoi(tokens["right"].c_str()); const int bottom = std::atoi(tokens["bottom"].c_str()); if (left < 0 || top < 0 || right <= left || bottom <= top || right > image.w || bottom > image.h) return nullptr; const int width = right - left, height = bottom - top; std::vector pixels(std::size_t(width) * height * 4); for (int y = 0; y < height; ++y) std::memcpy(pixels.data() + std::size_t(y) * width * 4, image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4, std::size_t(width) * 4); SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32, pixels.data(), width * 4); if (!surface) return nullptr; const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0; const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0; SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y); SDL_DestroySurface(surface); return cursor; } class NativeAudioEngine { struct MemoryAudioBuffer { const std::uint8_t* data = nullptr; std::size_t size = 0; }; struct Voice { const std::vector* samples = nullptr; std::size_t frame_cursor = 0; float volume = 1.0f; float target_volume = 1.0f; float fade_step_per_frame = 0.0f; bool stop_after_fade = false; bool loop = false; bool is_3d = false; int handle = 0; std::string filename; }; #ifdef __APPLE__ static OSStatus mem_audio_read_proc( void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) { const auto* mem = static_cast(inClientData); if (inPosition < 0 || static_cast(inPosition) >= mem->size) { *actualCount = 0; return noErr; } const std::size_t avail = mem->size - static_cast(inPosition); const std::size_t to_read = std::min(requestCount, avail); std::memcpy(buffer, mem->data + inPosition, to_read); *actualCount = static_cast(to_read); return noErr; } static SInt64 mem_audio_get_size_proc(void* inClientData) { return static_cast(static_cast(inClientData)->size); } #endif public: NativeAudioEngine() { if (SDL_InitSubSystem(SDL_INIT_AUDIO)) { SDL_AudioSpec spec{}; spec.format = SDL_AUDIO_F32; spec.channels = 2; spec.freq = 44100; stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_PLAYBACK, &spec, nullptr, nullptr); if (stream_) SDL_ResumeAudioStreamDevice(stream_); } // Always DrainAudioCommands() from cold boot so stale queued commands don't accumulate. (void)DrainAudioCommands(); } ~NativeAudioEngine() { if (stream_) SDL_DestroyAudioStream(stream_); SDL_QuitSubSystem(SDL_INIT_AUDIO); } void pump() { const auto commands = DrainAudioCommands(); for (const auto& cmd : commands) { ++commands_drained_; switch (cmd.type) { case AudioCommand::PlaySound2D: { const auto* clip = get_clip(cmd.filename); if (clip && !clip->empty()) { Voice v{}; v.samples = clip; v.volume = sound_volume_; v.target_volume = sound_volume_; v.filename = cmd.filename; voices_.push_back(std::move(v)); ++played_2d_; } break; } case AudioCommand::PlaySound3D: { const auto* clip = get_clip(cmd.filename); if (clip && !clip->empty()) { Voice v{}; v.samples = clip; v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); v.target_volume = v.volume; v.loop = cmd.play_count != 1; v.is_3d = true; v.handle = cmd.id; v.filename = cmd.filename; voices_.push_back(std::move(v)); ++played_3d_; } break; } case AudioCommand::StopSound3D: voices_.erase( std::remove_if(voices_.begin(), voices_.end(), [&](const Voice& v) { return v.is_3d && v.handle == cmd.id; }), voices_.end()); break; case AudioCommand::StopAllSound3D: voices_.erase( std::remove_if(voices_.begin(), voices_.end(), [](const Voice& v) { return v.is_3d; }), voices_.end()); break; case AudioCommand::SetSoundVolume3D: for (auto& v : voices_) { if (v.is_3d && v.handle == cmd.id) { v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); v.target_volume = v.volume; } } break; case AudioCommand::PlayMusic: case AudioCommand::FadeInMusic: { const auto* clip = get_clip(cmd.filename); if (clip && !clip->empty()) { bgm_.samples = clip; bgm_.frame_cursor = 0; bgm_.loop = true; bgm_.filename = cmd.filename; bgm_.stop_after_fade = false; const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); bgm_.target_volume = target; if (cmd.type == AudioCommand::FadeInMusic) { bgm_.volume = 0.0f; bgm_.fade_step_per_frame = target / (44100.0f * 1.5f); } else { bgm_.volume = target; bgm_.fade_step_per_frame = 0.0f; } ++played_music_; } break; } case AudioCommand::FadeOutMusic: case AudioCommand::FadeOutAllMusic: if (bgm_.samples) { bgm_.target_volume = 0.0f; bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f); bgm_.stop_after_fade = true; } break; case AudioCommand::FadeLimitOutMusic: if (bgm_.samples) { bgm_.target_volume = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); bgm_.fade_step_per_frame = (bgm_.target_volume - bgm_.volume) / (44100.0f * 1.0f); bgm_.stop_after_fade = false; } break; case AudioCommand::SetMusicVolume: music_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); if (bgm_.samples && !bgm_.stop_after_fade) { bgm_.volume = music_volume_; bgm_.target_volume = music_volume_; } break; case AudioCommand::SetSoundVolume: sound_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); break; default: break; } } if (!stream_) return; const int queued_bytes = SDL_GetAudioStreamAvailable(stream_); if (queued_bytes < 0) return; const std::size_t queued_frames = static_cast(queued_bytes) / (2 * sizeof(float)); constexpr std::size_t kTargetQueuedFrames = 4410; // ~100 ms stereo @ 44.1 kHz if (queued_frames >= kTargetQueuedFrames) return; const std::size_t frames_to_mix = kTargetQueuedFrames - queued_frames; mix_buffer_.assign(frames_to_mix * 2, 0.0f); auto mix_voice = [&](Voice& v) -> bool { if (!v.samples || v.samples->empty()) return false; const std::size_t total_frames = v.samples->size() / 2; if (total_frames == 0) return false; const float* src = v.samples->data(); for (std::size_t f = 0; f < frames_to_mix; ++f) { if (v.frame_cursor >= total_frames) { if (v.loop) v.frame_cursor = 0; else return false; } if (v.fade_step_per_frame != 0.0f) { v.volume += v.fade_step_per_frame; if ((v.fade_step_per_frame > 0.0f && v.volume >= v.target_volume) || (v.fade_step_per_frame < 0.0f && v.volume <= v.target_volume)) { v.volume = v.target_volume; v.fade_step_per_frame = 0.0f; if (v.stop_after_fade && v.volume <= 0.0001f) { v.samples = nullptr; return false; } } } mix_buffer_[f * 2 + 0] += src[v.frame_cursor * 2 + 0] * v.volume; mix_buffer_[f * 2 + 1] += src[v.frame_cursor * 2 + 1] * v.volume; ++v.frame_cursor; } return v.loop || v.frame_cursor < total_frames; }; if (bgm_.samples) { if (!mix_voice(bgm_)) bgm_.samples = nullptr; } for (auto it = voices_.begin(); it != voices_.end();) { if (!mix_voice(*it)) it = voices_.erase(it); else ++it; } for (float& sample : mix_buffer_) sample = std::clamp(sample, -1.0f, 1.0f); SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast(mix_buffer_.size() * sizeof(float))); } private: const std::vector* get_clip(const std::string& vpath) { if (vpath.empty()) return nullptr; auto [it, inserted] = clips_.try_emplace(vpath); if (!inserted) return &it->second; #ifdef __APPLE__ std::vector bytes; if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16) return &it->second; MemoryAudioBuffer mem{bytes.data(), bytes.size()}; AudioFileID audio_file = nullptr; if (AudioFileOpenWithCallbacks( &mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr || !audio_file) { return &it->second; } ExtAudioFileRef ext_file = nullptr; if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) { AudioFileClose(audio_file); return &it->second; } AudioStreamBasicDescription client_format{}; client_format.mSampleRate = 44100.0; client_format.mFormatID = kAudioFormatLinearPCM; client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked; client_format.mBytesPerPacket = 8; client_format.mFramesPerPacket = 1; client_format.mBytesPerFrame = 8; client_format.mChannelsPerFrame = 2; client_format.mBitsPerChannel = 32; if (ExtAudioFileSetProperty( ext_file, kExtAudioFileProperty_ClientDataFormat, sizeof(client_format), &client_format) == noErr) { std::vector chunk(4096 * 2); while (true) { UInt32 frame_count = 4096; AudioBufferList buf_list{}; buf_list.mNumberBuffers = 1; buf_list.mBuffers[0].mNumberChannels = 2; buf_list.mBuffers[0].mDataByteSize = static_cast(chunk.size() * sizeof(float)); buf_list.mBuffers[0].mData = chunk.data(); if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0) break; it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2); if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes break; } } ExtAudioFileDispose(ext_file); AudioFileClose(audio_file); #endif return &it->second; } SDL_AudioStream* stream_ = nullptr; std::unordered_map> clips_; std::vector voices_; Voice bgm_{}; std::vector mix_buffer_; float sound_volume_ = 0.8f; float music_volume_ = 0.5f; std::size_t commands_drained_ = 0; std::size_t played_2d_ = 0; std::size_t played_3d_ = 0; std::size_t played_music_ = 0; }; #endif std::string bundled_file_path(const char* name) { const char* base = SDL_GetBasePath(); return std::string(base ? base : "./") + name; } std::vector read_spirv(const char* name) { std::size_t size = 0; void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size); #ifdef __ANDROID__ if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size); if (!bytes) bytes = SDL_LoadFile(name, &size); #endif if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name); if (size == 0 || size % 4) { SDL_free(bytes); throw std::runtime_error(std::string("invalid SPIR-V size: ") + name); } std::vector words(size / 4); std::memcpy(words.data(), bytes, size); SDL_free(bytes); return words; } class VulkanWindow { struct GeometryId { std::uint64_t key = 0, signature = 0; bool operator==(const GeometryId&) const = default; }; struct GeometryIdHash { std::size_t operator()(GeometryId id) const { return std::size_t(id.key ^ (id.signature + 0x9e3779b97f4a7c15ull + (id.key << 6) + (id.key >> 2))); } }; struct Geometry { VkBuffer buffer = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; VkDeviceSize index_offset = 0; std::uint64_t last_used_frame = 0; std::uint32_t vertex_count = 0, index_count = 0; }; struct GpuTexture { VkImage image = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; VkImageView view = VK_NULL_HANDLE; VkDescriptorSet descriptor = VK_NULL_HANDLE; VkDescriptorSet ui_descriptor = VK_NULL_HANDLE; std::uint64_t last_used_frame = 0; std::uint64_t last_decode_attempt_frame = 0; }; struct PairedDescriptor { VkDescriptorSet set = VK_NULL_HANDLE; std::uint64_t last_used_frame = 0; }; struct PreparedDraw { Geometry* geometry = nullptr; VkPipeline pipeline = VK_NULL_HANDLE; VkDescriptorSet descriptor = VK_NULL_HANDLE; PushConstants constants{}; FixedFunctionState fixed_state{}; std::uint32_t state_offset = 0; }; struct UiBatch { std::uint32_t first_index = 0; std::uint32_t index_count = 0; VkPipeline pipeline = VK_NULL_HANDLE; VkDescriptorSet descriptor = VK_NULL_HANDLE; PushConstants constants{}; bool behind_3d = false; std::uint32_t state_offset = 0; }; static constexpr std::size_t kMaxBonesPerFrame = 65536; static constexpr VkDeviceSize kBoneBufferBytes = kMaxBonesPerFrame * 16 * sizeof(float); static constexpr std::size_t kMaxUiVerticesPerFrame = 65536; static constexpr std::size_t kMaxUiIndicesPerFrame = 98304; static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex); static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t); public: struct Timings { double sync_ms = 0; double prepare_ms = 0; double steady_prepare_ms = 0; double submit_ms = 0; double present_ms = 0; double gpu_ms = 0; std::uint64_t gpu_samples = 0; }; #ifdef MT_NATIVE_HAS_LIVE_CLIENT void enable_game_hardware_cursor() { hardware_cursor_enabled_ = true; for (int shape = 0; shape < static_cast(game_cursors_.size()); ++shape) { game_cursors_[shape] = load_game_cursor(shape); } fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT); SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_); SDL_ShowCursor(); os_cursor_hidden_ = false; } void sync_game_cursor() { if (!hardware_cursor_enabled_) return; const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() && !PythonBoot::IsSoftwareCursorVisible(); if (visible == os_cursor_hidden_) { if (visible) SDL_ShowCursor(); else SDL_HideCursor(); os_cursor_hidden_ = !visible; } if (!visible) return; const int shape = PythonBoot::CursorShape(); if (shape == current_cursor_shape_) return; SDL_Cursor* cursor = shape >= 0 && shape < static_cast(game_cursors_.size()) ? game_cursors_[shape] : nullptr; if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_); current_cursor_shape_ = shape; } #endif explicit VulkanWindow(bool vsync = true, int init_width = 960, int init_height = 640) : vsync_(vsync) { #ifdef __APPLE__ if (!std::getenv("VK_ICD_FILENAMES")) { for (const char* icd : { "/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json", "/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json"}) { if (std::ifstream(icd).good()) { setenv("VK_ICD_FILENAMES", icd, 0); break; } } } #endif if (!SDL_Init(SDL_INIT_VIDEO)) throw std::runtime_error(SDL_GetError()); window_ = SDL_CreateWindow( "Metin2 Native Vulkan Client", std::max(init_width, 320), std::max(init_height, 240), SDL_WINDOW_VULKAN | SDL_WINDOW_RESIZABLE | SDL_WINDOW_HIGH_PIXEL_DENSITY); if (!window_) throw std::runtime_error(SDL_GetError()); SDL_StartTextInput(window_); Uint32 extension_count = 0; const char* const* sdl_extensions = SDL_Vulkan_GetInstanceExtensions(&extension_count); if (!sdl_extensions) throw std::runtime_error(SDL_GetError()); std::vector extensions(sdl_extensions, sdl_extensions + extension_count); std::uint32_t available_count = 0; check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, nullptr), "vkEnumerateInstanceExtensionProperties count"); std::vector available(available_count); check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, available.data()), "vkEnumerateInstanceExtensionProperties"); const bool portability = std::any_of(available.begin(), available.end(), [](const auto& extension) { return std::strcmp(extension.extensionName, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; }); if (portability && std::find_if(extensions.begin(), extensions.end(), [](const char* name) { return std::strcmp(name, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; }) == extensions.end()) extensions.push_back(VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME); VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO}; app.pApplicationName = "Metin2 native renderer"; app.apiVersion = VK_API_VERSION_1_1; VkInstanceCreateInfo instance_info{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO}; instance_info.flags = portability ? VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR : 0; instance_info.pApplicationInfo = &app; instance_info.enabledExtensionCount = static_cast(extensions.size()); instance_info.ppEnabledExtensionNames = extensions.data(); check(vkCreateInstance(&instance_info, nullptr, &instance_), "vkCreateInstance"); if (!SDL_Vulkan_CreateSurface(window_, instance_, nullptr, &surface_)) throw std::runtime_error(SDL_GetError()); select_device(); VkCommandPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; pool_info.queueFamilyIndex = queue_family_; check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool"); VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1; check(vkAllocateCommandBuffers(device_, &alloc, &command_), "vkAllocateCommandBuffers"); VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO}; query_info.queryType = VK_QUERY_TYPE_TIMESTAMP; query_info.queryCount = 2; check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool"); create_swapchain(); create_descriptors_and_buffers(); create_render_pass_and_layout(); VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &acquire_), "vkCreateSemaphore acquire"); check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &rendered_), "vkCreateSemaphore rendered"); VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; check(vkCreateFence(device_, &fence_info, nullptr, &fence_), "vkCreateFence"); touch_controller.update_screen_size(int(width()), int(height())); } ~VulkanWindow() { if (hardware_cursor_enabled_) { SDL_SetCursor(nullptr); for (SDL_Cursor* cursor : game_cursors_) if (cursor) SDL_DestroyCursor(cursor); if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_); } if (device_) vkDeviceWaitIdle(device_); if (bone_mapped_) vkUnmapMemory(device_, bone_memory_); if (bone_buffer_) vkDestroyBuffer(device_, bone_buffer_, nullptr); if (bone_memory_) vkFreeMemory(device_, bone_memory_, nullptr); if (state_mapped_) vkUnmapMemory(device_, state_memory_); if (state_buffer_) vkDestroyBuffer(device_, state_buffer_, nullptr); if (state_memory_) vkFreeMemory(device_, state_memory_, nullptr); if (ui_mapped_) vkUnmapMemory(device_, ui_memory_); if (ui_buffer_) vkDestroyBuffer(device_, ui_buffer_, nullptr); if (ui_memory_) vkFreeMemory(device_, ui_memory_, nullptr); if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr); if (fence_) vkDestroyFence(device_, fence_, nullptr); if (acquire_) vkDestroySemaphore(device_, acquire_, nullptr); if (rendered_) vkDestroySemaphore(device_, rendered_, nullptr); for (auto& entry : geometries_) release_geometry(entry.second); for (auto& entry : textures_) release_texture(entry.second); release_texture(fallback_texture_); for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr); if (fragment_module_) vkDestroyShaderModule(device_, fragment_module_, nullptr); if (layout_) vkDestroyPipelineLayout(device_, layout_, nullptr); if (descriptor_pool_) vkDestroyDescriptorPool(device_, descriptor_pool_, nullptr); if (descriptor_layout_) vkDestroyDescriptorSetLayout(device_, descriptor_layout_, nullptr); if (bone_descriptor_layout_) vkDestroyDescriptorSetLayout(device_, bone_descriptor_layout_, nullptr); if (sampler_) vkDestroySampler(device_, sampler_, nullptr); if (clamp_sampler_) vkDestroySampler(device_, clamp_sampler_, nullptr); if (ui_sampler_) vkDestroySampler(device_, ui_sampler_, nullptr); for (const auto& entry : state_samplers_) vkDestroySampler(device_, entry.second, nullptr); for (auto framebuffer : framebuffers_) vkDestroyFramebuffer(device_, framebuffer, nullptr); if (pass_) vkDestroyRenderPass(device_, pass_, nullptr); if (depth_view_) vkDestroyImageView(device_, depth_view_, nullptr); if (depth_image_) vkDestroyImage(device_, depth_image_, nullptr); if (depth_memory_) vkFreeMemory(device_, depth_memory_, nullptr); for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); if (swapchain_) vkDestroySwapchainKHR(device_, swapchain_, nullptr); if (pool_) vkDestroyCommandPool(device_, pool_, nullptr); if (device_) vkDestroyDevice(device_, nullptr); if (surface_) vkDestroySurfaceKHR(instance_, surface_, nullptr); if (instance_) vkDestroyInstance(instance_, nullptr); if (window_) { SDL_StopTextInput(window_); SDL_DestroyWindow(window_); } SDL_Quit(); } std::uint32_t width() const { return extent_.width; } std::uint32_t height() const { return extent_.height; } std::uint32_t logical_width() const { int w = 0, h = 0; if (window_) SDL_GetWindowSize(window_, &w, &h); return w > 0 ? static_cast(w) : extent_.width; } std::uint32_t logical_height() const { int w = 0, h = 0; if (window_) SDL_GetWindowSize(window_, &w, &h); return h > 0 ? static_cast(h) : extent_.height; } TouchController touch_controller; static int sdl_scancode_to_dik(SDL_Scancode sc) { switch (sc) { case SDL_SCANCODE_ESCAPE: return 0x01; case SDL_SCANCODE_1: return 0x02; case SDL_SCANCODE_2: return 0x03; case SDL_SCANCODE_3: return 0x04; case SDL_SCANCODE_4: return 0x05; case SDL_SCANCODE_5: return 0x06; case SDL_SCANCODE_6: return 0x07; case SDL_SCANCODE_7: return 0x08; case SDL_SCANCODE_8: return 0x09; case SDL_SCANCODE_9: return 0x0A; case SDL_SCANCODE_0: return 0x0B; case SDL_SCANCODE_MINUS: return 0x0C; case SDL_SCANCODE_EQUALS: return 0x0D; case SDL_SCANCODE_BACKSPACE: return 0x0E; case SDL_SCANCODE_TAB: return 0x0F; case SDL_SCANCODE_Q: return 0x10; case SDL_SCANCODE_W: return 0x11; case SDL_SCANCODE_E: return 0x12; case SDL_SCANCODE_R: return 0x13; case SDL_SCANCODE_T: return 0x14; case SDL_SCANCODE_Y: return 0x15; case SDL_SCANCODE_U: return 0x16; case SDL_SCANCODE_I: return 0x17; case SDL_SCANCODE_O: return 0x18; case SDL_SCANCODE_P: return 0x19; case SDL_SCANCODE_LEFTBRACKET: return 0x1A; case SDL_SCANCODE_RIGHTBRACKET: return 0x1B; case SDL_SCANCODE_RETURN: return 0x1C; case SDL_SCANCODE_LCTRL: return 0x1D; case SDL_SCANCODE_A: return 0x1E; case SDL_SCANCODE_S: return 0x1F; case SDL_SCANCODE_D: return 0x20; case SDL_SCANCODE_F: return 0x21; case SDL_SCANCODE_G: return 0x22; case SDL_SCANCODE_H: return 0x23; case SDL_SCANCODE_J: return 0x24; case SDL_SCANCODE_K: return 0x25; case SDL_SCANCODE_L: return 0x26; case SDL_SCANCODE_SEMICOLON: return 0x27; case SDL_SCANCODE_APOSTROPHE: return 0x28; case SDL_SCANCODE_GRAVE: return 0x29; case SDL_SCANCODE_LSHIFT: return 0x2A; case SDL_SCANCODE_BACKSLASH: return 0x2B; case SDL_SCANCODE_Z: return 0x2C; case SDL_SCANCODE_X: return 0x2D; case SDL_SCANCODE_C: return 0x2E; case SDL_SCANCODE_V: return 0x2F; case SDL_SCANCODE_B: return 0x30; case SDL_SCANCODE_N: return 0x31; case SDL_SCANCODE_M: return 0x32; case SDL_SCANCODE_COMMA: return 0x33; case SDL_SCANCODE_PERIOD: return 0x34; case SDL_SCANCODE_SLASH: return 0x35; case SDL_SCANCODE_RSHIFT: return 0x36; case SDL_SCANCODE_KP_MULTIPLY: return 0x37; case SDL_SCANCODE_LALT: return 0x38; case SDL_SCANCODE_SPACE: return 0x39; case SDL_SCANCODE_CAPSLOCK: return 0x3A; case SDL_SCANCODE_F1: return 0x3B; case SDL_SCANCODE_F2: return 0x3C; case SDL_SCANCODE_F3: return 0x3D; case SDL_SCANCODE_F4: return 0x3E; case SDL_SCANCODE_F5: return 0x3F; case SDL_SCANCODE_F6: return 0x40; case SDL_SCANCODE_F7: return 0x41; case SDL_SCANCODE_F8: return 0x42; case SDL_SCANCODE_F9: return 0x43; case SDL_SCANCODE_F10: return 0x44; case SDL_SCANCODE_NUMLOCKCLEAR: return 0x45; case SDL_SCANCODE_SCROLLLOCK: return 0x46; case SDL_SCANCODE_KP_7: return 0x47; case SDL_SCANCODE_KP_8: return 0x48; case SDL_SCANCODE_KP_9: return 0x49; case SDL_SCANCODE_KP_MINUS: return 0x4A; case SDL_SCANCODE_KP_4: return 0x4B; case SDL_SCANCODE_KP_5: return 0x4C; case SDL_SCANCODE_KP_6: return 0x4D; case SDL_SCANCODE_KP_PLUS: return 0x4E; case SDL_SCANCODE_KP_1: return 0x4F; case SDL_SCANCODE_KP_2: return 0x50; case SDL_SCANCODE_KP_3: return 0x51; case SDL_SCANCODE_KP_0: return 0x52; case SDL_SCANCODE_KP_PERIOD: return 0x53; case SDL_SCANCODE_F11: return 0x57; case SDL_SCANCODE_F12: return 0x58; case SDL_SCANCODE_KP_ENTER: return 0x9C; case SDL_SCANCODE_RCTRL: return 0x9D; case SDL_SCANCODE_KP_DIVIDE: return 0xB5; case SDL_SCANCODE_RALT: return 0xB8; case SDL_SCANCODE_HOME: return 0xC7; case SDL_SCANCODE_UP: return 0xC8; case SDL_SCANCODE_PAGEUP: return 0xC9; case SDL_SCANCODE_LEFT: return 0xCB; case SDL_SCANCODE_RIGHT: return 0xCD; case SDL_SCANCODE_END: return 0xCF; case SDL_SCANCODE_DOWN: return 0xD0; case SDL_SCANCODE_PAGEDOWN: return 0xD1; case SDL_SCANCODE_INSERT: return 0xD2; case SDL_SCANCODE_DELETE: return 0xD3; default: return 0; } } static int sdl_scancode_to_vk(SDL_Scancode sc) { switch (sc) { case SDL_SCANCODE_BACKSPACE: return 0x08; case SDL_SCANCODE_TAB: return 0x09; case SDL_SCANCODE_RETURN: case SDL_SCANCODE_KP_ENTER: return 0x0D; case SDL_SCANCODE_ESCAPE: return 0x1B; case SDL_SCANCODE_SPACE: return 0x20; case SDL_SCANCODE_PAGEUP: return 0x21; case SDL_SCANCODE_PAGEDOWN: return 0x22; case SDL_SCANCODE_END: return 0x23; case SDL_SCANCODE_HOME: return 0x24; case SDL_SCANCODE_LEFT: return 0x25; case SDL_SCANCODE_UP: return 0x26; case SDL_SCANCODE_RIGHT: return 0x27; case SDL_SCANCODE_DOWN: return 0x28; case SDL_SCANCODE_INSERT: return 0x2D; case SDL_SCANCODE_DELETE: return 0x2E; case SDL_SCANCODE_F1: return 0x70; case SDL_SCANCODE_F2: return 0x71; case SDL_SCANCODE_F3: return 0x72; case SDL_SCANCODE_F4: return 0x73; case SDL_SCANCODE_F5: return 0x74; case SDL_SCANCODE_F6: return 0x75; case SDL_SCANCODE_F7: return 0x76; case SDL_SCANCODE_F8: return 0x77; case SDL_SCANCODE_F9: return 0x78; case SDL_SCANCODE_F10: return 0x79; case SDL_SCANCODE_F11: return 0x7A; case SDL_SCANCODE_F12: return 0x7B; default: return 0; } } bool poll(bool forward_to_live_client = false) { #ifdef MT_NATIVE_HAS_LIVE_CLIENT auto map_mouse = [&](float wx, float wy, int& ox, int& oy) { int win_w = 0, win_h = 0; SDL_GetWindowSize(window_, &win_w, &win_h); unsigned ui_w = extent_.width ? extent_.width : 960; unsigned ui_h = extent_.height ? extent_.height : 640; UIRenderGetSize(&ui_w, &ui_h); if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) { ox = static_cast(std::lround(wx * float(ui_w) / float(win_w))); oy = static_cast(std::lround(wy * float(ui_h) / float(win_h))); } else { ox = static_cast(std::lround(wx)); oy = static_cast(std::lround(wy)); } }; std::optional> pending_mouse_move; auto flush_mouse_move = [&] { if (!pending_mouse_move) return; const auto [mx, my] = *pending_mouse_move; pending_mouse_move.reset(); last_mx_ = mx; last_my_ = my; if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return; PythonBoot::UIMouseMove(mx, my); }; #endif SDL_Event event; while (SDL_PollEvent(&event)) { #ifdef MT_NATIVE_HAS_LIVE_CLIENT if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) { int mx = 0, my = 0; map_mouse(event.motion.x, event.motion.y, mx, my); pending_mouse_move = std::pair{mx, my}; continue; } // Preserve event order at button/key/focus boundaries while collapsing // high-frequency motion into the most recent position. flush_mouse_move(); #endif if (event.type == SDL_EVENT_QUIT) return false; if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) { int ew = 0, eh = 0; SDL_GetWindowSize(window_, &ew, &eh); touch_controller.update_screen_size(ew, eh); #ifdef MT_NATIVE_HAS_LIVE_CLIENT if (forward_to_live_client && ew > 0 && eh > 0) { PythonBoot::SetUISize(ew, eh); } #endif recreate_swapchain(); continue; } if (event.type == SDL_EVENT_FINGER_DOWN) { touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); continue; } else if (event.type == SDL_EVENT_FINGER_MOTION) { touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); continue; } else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) { touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); continue; } #ifdef MT_NATIVE_HAS_LIVE_CLIENT if (forward_to_live_client) { if (!hardware_cursor_enabled_ && (event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) { std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n"); SDL_HideCursor(); os_cursor_hidden_ = true; } else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) { SDL_CaptureMouse(false); PythonBoot::UIMouseButton(2, false, last_mx_, last_my_); PythonBoot::UIMouseButton(3, false, last_mx_, last_my_); PythonBoot::UIMouseButton(1, false, last_mx_, last_my_); } if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) { const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN; int btn = 0; if (event.button.button == SDL_BUTTON_LEFT) btn = 1; else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2; else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3; if (btn) { int mx = 0, my = 0; map_mouse(event.button.x, event.button.y, mx, my); last_mx_ = mx; last_my_ = my; if (touch_controller.is_enabled()) { if (touch_controller.on_mouse_button(btn, pressed, mx, my)) { continue; } } SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0); PythonBoot::UIMouseButton(btn, pressed, mx, my); } } else if (event.type == SDL_EVENT_MOUSE_WHEEL) { PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f)); } else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) { const bool pressed = event.type == SDL_EVENT_KEY_DOWN; if (pressed) { if (const int vk = sdl_scancode_to_vk(event.key.scancode)) PythonBoot::UIIMEKeyDown(vk); if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8); else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9); else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER) PythonBoot::UIChar(13); else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27); } if (!event.key.repeat) { if (const int dik = sdl_scancode_to_dik(event.key.scancode)) PythonBoot::UIKey(dik, pressed); } } else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) { const auto* s = reinterpret_cast(event.text.text); while (*s) { unsigned cp = 0; if (*s < 0x80u) { cp = *s++; } else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) { cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu); s += 2; } else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) { cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu); s += 3; } else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) { cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu); s += 4; } else { ++s; continue; } if (cp >= 32u && cp != 127u) PythonBoot::UIChar(cp); } } } #else (void)forward_to_live_client; #endif } #ifdef MT_NATIVE_HAS_LIVE_CLIENT flush_mouse_move(); #endif touch_controller.update(); #ifdef MT_NATIVE_HAS_LIVE_CLIENT if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) { SDL_HideCursor(); os_cursor_hidden_ = true; } #endif return true; } void reset_timings() { collect_pending_gpu_timestamp(); timings_ = {}; timed_frames_ = 0; upload_count_ = 0; uploaded_bytes_ = 0; texture_upload_count_ = 0; texture_uploaded_bytes_ = 0; } void finish_gpu_timings() { if (!has_pending_query_) return; check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences finish"); collect_pending_gpu_timestamp(); } void render( const std::vector& draws, const std::unordered_map>& capture_textures, std::uint32_t ui_width = 960, std::uint32_t ui_height = 640, const std::vector& ui_commands = {}) { if (extent_.width == 0 || extent_.height == 0) return; const auto start = std::chrono::steady_clock::now(); check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences"); collect_pending_gpu_timestamp(); prune_textures(); std::uint32_t image_index = 0; const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, acquire_, VK_NULL_HANDLE, &image_index); if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { recreate_swapchain(); return; } if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR"); const auto synchronized = std::chrono::steady_clock::now(); ++frame_number_; ++timed_frames_; std::vector prepared_draws; prepared_draws.reserve(draws.size()); std::unordered_set active_keys; std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; std::uint32_t bone_cursor = 0; std::size_t required_bones = 0; for (const auto& draw : draws) { if (!draw.bone_matrices.empty()) { if (draw.bone_matrices.size() % 16) throw std::runtime_error("invalid bone palette length"); required_bones += draw.bone_matrices.size() / 16; } } ensure_bone_capacity(required_bones); for (const auto& draw : draws) { if (draw.positions.empty() || draw.indices.empty()) continue; const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw); const GeometryId id{draw.geometry_key, signature}; auto [it, inserted] = geometries_.try_emplace(id); auto& geometry = it->second; if (inserted) { upload_geometry(draw, geometry); } else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) { throw std::runtime_error("geometry key/revision reused with a different size"); } geometry.last_used_frame = frame_number_; if (draw.geometry_key) active_keys.insert(draw.geometry_key); float skin_offset_encoded = 0.0f; if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && bone_mapped_) { const auto bone_count = static_cast(draw.bone_matrices.size() / 16); if (std::size_t(bone_cursor) + bone_count <= bone_capacity_) { std::memcpy( bone_mapped_ + std::size_t(bone_cursor) * 16, draw.bone_matrices.data(), draw.bone_matrices.size() * sizeof(float)); skin_offset_encoded = float(bone_cursor + 1u); bone_cursor += bone_count; ++skinned_draw_count; } else throw std::runtime_error("bone palette capacity exceeded"); } const bool is_specular = draw_is_specular_spheremap(draw); const bool is_alpha = draw.alpha_blend != 0; std::uint8_t cull = 0; // D3D8 cull mode names describe the winding to discard. With the // shader's Y flip, clockwise is Vulkan's front-facing winding. if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0; std::uint8_t blend = 0; if (is_alpha) { if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 && draw.dest_blend >= 1 && draw.dest_blend <= 11) { blend = static_cast((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4)); } else { blend = 1; } } const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, static_cast(draw.z_func)); const bool bind_tex1 = is_specular || (!draw.texture1.empty() && (draw.color_op[1] > 1 || draw.alpha_op[1] > 1)); const auto descriptor = get_texture_descriptor( draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw); auto constants = make_push_constants(draw, skin_offset_encoded); if (draw.pretransformed) { const float screen_width = float(logical_width()); const float screen_height = float(logical_height()); constants.mvp = {2.0f / screen_width, 0, 0, 0, 0, 2.0f / screen_height, 0, 0, 0, 0, 1, 0, -1, -1, 0, 1}; } prepared_draws.push_back({ &geometry, pipeline, descriptor, constants, make_fixed_function_state(draw)}); vertex_count += geometry.vertex_count; index_count += geometry.index_count; } for (auto it = geometries_.begin(); it != geometries_.end();) { const auto age = frame_number_ - it->second.last_used_frame; const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key); if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) { release_geometry(it->second); it = geometries_.erase(it); } else ++it; } std::vector ui_batches; std::size_t ui_quad_count = 0; if (ui_mapped_) { const uint32_t target_ui_w = ui_width ? ui_width : 960; const uint32_t target_ui_h = ui_height ? ui_height : 640; if (touch_controller.is_enabled()) { touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h)); std::vector combined_ui = ui_commands; touch_controller.append_ui_commands(combined_ui); if (!combined_ui.empty()) { build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count); } } else if (!ui_commands.empty()) { build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count); } } ensure_state_capacity(prepared_draws.size() + ui_batches.size()); std::size_t state_index = 0; for (auto& draw : prepared_draws) { draw.state_offset = static_cast(state_index * state_stride_); std::memcpy(state_mapped_ + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); ++state_index; } const FixedFunctionState ui_state{}; for (auto& batch : ui_batches) { batch.state_offset = static_cast(state_index * state_stride_); std::memcpy(state_mapped_ + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); ++state_index; } const auto prepared = std::chrono::steady_clock::now(); check(vkResetFences(device_, 1, &fence_), "vkResetFences"); check(vkResetCommandBuffer(command_, 0), "vkResetCommandBuffer"); VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; check(vkBeginCommandBuffer(command_, &begin), "vkBeginCommandBuffer"); vkCmdResetQueryPool(command_, query_pool_, 0, 2); vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, 0); VkClearValue clears[2]{}; clears[0].color = {{0.035f, 0.05f, 0.075f, 1.0f}}; clears[1].depthStencil = {1.0f, 0}; VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = 2; pass_begin.pClearValues = clears; vkCmdBeginRenderPass(command_, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); VkPipeline current_pipeline = VK_NULL_HANDLE; VkDescriptorSet current_descriptor = VK_NULL_HANDLE; auto record_ui_pass = [&](bool behind_3d) { bool ui_bound = false; for (const auto& batch : ui_batches) { if (batch.behind_3d != behind_3d) continue; if (!ui_bound) { const VkDeviceSize v_offset = 0; vkCmdBindVertexBuffers(command_, 0, 1, &ui_buffer_, &v_offset); vkCmdBindIndexBuffer(command_, ui_buffer_, ui_vertex_bytes_, VK_INDEX_TYPE_UINT32); ui_bound = true; } if (batch.pipeline != current_pipeline) { vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); current_pipeline = batch.pipeline; } if (batch.descriptor != current_descriptor) { vkCmdBindDescriptorSets( command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); current_descriptor = batch.descriptor; } vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, &bone_descriptor_set_, 1, &batch.state_offset); vkCmdPushConstants( command_, layout_, VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, 0, sizeof(batch.constants), &batch.constants); vkCmdDrawIndexed(command_, batch.index_count, 1, batch.first_index, 0, 0); } }; record_ui_pass(true); for (const auto& prepared_draw : prepared_draws) { if (prepared_draw.pipeline != current_pipeline) { vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); current_pipeline = prepared_draw.pipeline; } if (prepared_draw.descriptor != current_descriptor) { vkCmdBindDescriptorSets( command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); current_descriptor = prepared_draw.descriptor; } vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, &bone_descriptor_set_, 1, &prepared_draw.state_offset); const auto& geometry = *prepared_draw.geometry; const VkDeviceSize vertex_offset = 0; vkCmdBindVertexBuffers(command_, 0, 1, &geometry.buffer, &vertex_offset); vkCmdBindIndexBuffer(command_, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); vkCmdPushConstants( command_, layout_, VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, 0, sizeof(prepared_draw.constants), &prepared_draw.constants); vkCmdDrawIndexed(command_, geometry.index_count, 1, 0, 0, 0); } record_ui_pass(false); vkCmdEndRenderPass(command_); vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, 1); check(vkEndCommandBuffer(command_), "vkEndCommandBuffer"); has_pending_query_ = true; const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &acquire_; submit.pWaitDstStageMask = &stage; submit.commandBufferCount = 1; submit.pCommandBuffers = &command_; submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_; check(vkQueueSubmit(queue_, 1, &submit, fence_), "vkQueueSubmit"); const auto submitted = std::chrono::steady_clock::now(); VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_; present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; const auto presented = vkQueuePresentKHR(queue_, &present); if (presented == VK_ERROR_OUT_OF_DATE_KHR || presented == VK_SUBOPTIMAL_KHR) { recreate_swapchain(); } else if (presented != VK_SUCCESS) { check(presented, "vkQueuePresentKHR"); } const auto presented_at = std::chrono::steady_clock::now(); const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); timings_.sync_ms += std::chrono::duration(synchronized - start).count(); timings_.prepare_ms += prepare_ms; if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); timings_.present_ms += std::chrono::duration(presented_at - submitted).count(); last_draw_count_ = prepared_draws.size(); last_skinned_draw_count_ = skinned_draw_count; last_ui_batch_count_ = ui_batches.size(); last_ui_quad_count_ = ui_quad_count; last_vertex_count_ = vertex_count; last_index_count_ = index_count; } std::size_t draw_count() const { return last_draw_count_; } std::size_t skinned_draw_count() const { return last_skinned_draw_count_; } std::size_t ui_batch_count() const { return last_ui_batch_count_; } std::size_t ui_quad_count() const { return last_ui_quad_count_; } std::size_t vertex_count() const { return last_vertex_count_; } std::size_t index_count() const { return last_index_count_; } std::size_t upload_count() const { return upload_count_; } std::size_t uploaded_bytes() const { return uploaded_bytes_; } std::size_t texture_upload_count() const { return texture_upload_count_; } std::size_t texture_uploaded_bytes() const { return texture_uploaded_bytes_; } const std::string& device_name() const { return device_name_; } const char* present_mode_name() const { switch (present_mode_) { case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE"; case VK_PRESENT_MODE_MAILBOX_KHR: return "MAILBOX"; default: return "FIFO"; } } Timings timings() const { return timings_; } private: void recreate_swapchain() { int w = 0, h = 0; SDL_GetWindowSizeInPixels(window_, &w, &h); if (w <= 0 || h <= 0) return; vkDeviceWaitIdle(device_); for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); pipelines_.clear(); for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr); framebuffers_.clear(); if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; } if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; } if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; } for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); image_views_.clear(); if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; } create_swapchain(); for (auto view : image_views_) { VkImageView fb_attachments[2] = {view, depth_view_}; VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; framebuffer.renderPass = pass_; framebuffer.attachmentCount = 2; framebuffer.pAttachments = fb_attachments; framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; VkFramebuffer handle = VK_NULL_HANDLE; check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer recreate"); framebuffers_.push_back(handle); } get_pipeline(0, 2, 0); } void collect_pending_gpu_timestamp() { if (!has_pending_query_) return; has_pending_query_ = false; std::uint64_t timestamps[2] = {}; const VkResult res = vkGetQueryPoolResults( device_, query_pool_, 0, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), VK_QUERY_RESULT_64_BIT); if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) { const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_); timings_.gpu_ms += ns * 1e-6; ++timings_.gpu_samples; } } void build_ui_batches( std::uint32_t ui_width, std::uint32_t ui_height, const std::vector& commands, const std::unordered_map>& capture_textures, std::vector& batches, std::size_t& out_quad_count) { ensure_ui_capacity(commands.size()); auto* vertices = reinterpret_cast(ui_mapped_); auto* indices = reinterpret_cast(ui_mapped_ + ui_vertex_bytes_); std::uint32_t v_count = 0; std::uint32_t i_count = 0; const float inv_w = 2.0f / float(ui_width); const float inv_h = 2.0f / float(ui_height); for (const auto& cmd : commands) { if (v_count + 4 > ui_vertex_capacity_ || i_count + 6 > ui_index_capacity_) throw std::runtime_error("UI command buffer capacity exceeded"); if (cmd.kind == UIRenderCommand::Text) continue; float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2; float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev; const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1; float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; std::array top_color = unpack_argb(cmd.argb); std::array bot_color = cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color; if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue; if (cmd.kind == UIRenderCommand::Line) { const float dx = x2 - x1, dy = y2 - y1; const float len = std::sqrt(dx * dx + dy * dy); if (len < 0.1f) continue; const float nx = -dy / len * 0.5f; const float ny = dx / len * 0.5f; qx[0] = x1 + nx; qy[0] = y1 + ny; qx[1] = x2 + nx; qy[1] = y2 + ny; qx[2] = x1 - nx; qy[2] = y1 - ny; qx[3] = x2 - nx; qy[3] = y2 - ny; for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; } } else if (cmd.kind == UIRenderCommand::Image && cmd.quad) { const bool axis_aligned = std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f && std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f && std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f && std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f; for (int k = 0; k < 4; ++k) { mu[k] = cmd.mu[k]; mv[k] = cmd.mv[k]; } if (has_clip && axis_aligned) { const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3]; const float cx1 = std::max(ox1, cmd.clip_x1); const float cy1 = std::max(oy1, cmd.clip_y1); const float cx2 = std::min(ox2, cmd.clip_x2); const float cy2 = std::min(oy2, cmd.clip_y2); if (cx2 <= cx1 || cy2 <= cy1) continue; float msu = cmd.mu[0], meu = cmd.mu[1]; float msv = cmd.mv[0], mev = cmd.mv[2]; if (ox2 > ox1) { const float t0 = (cx1 - ox1) / (ox2 - ox1); const float t1 = (cx2 - ox1) / (ox2 - ox1); const float du = eu - su; su = cmd.su + du * t0; eu = cmd.su + du * t1; const float mdu = meu - msu; msu = cmd.mu[0] + mdu * t0; meu = cmd.mu[0] + mdu * t1; } if (oy2 > oy1) { const float t0 = (cy1 - oy1) / (oy2 - oy1); const float t1 = (cy2 - oy1) / (oy2 - oy1); const float dv = ev - sv; sv = cmd.sv + dv * t0; ev = cmd.sv + dv * t1; const float mdv = mev - msv; msv = cmd.mv[0] + mdv * t0; mev = cmd.mv[0] + mdv * t1; } qx[0] = cx1; qy[0] = cy1; qx[1] = cx2; qy[1] = cy1; qx[2] = cx1; qy[2] = cy2; qx[3] = cx2; qy[3] = cy2; mu[0] = msu; mv[0] = msv; mu[1] = meu; mv[1] = msv; mu[2] = msu; mv[2] = mev; mu[3] = meu; mv[3] = mev; } else { if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) continue; for (int k = 0; k < 4; ++k) { qx[k] = cmd.qx[k]; qy[k] = cmd.qy[k]; } } u[0] = su; v[0] = sv; u[1] = eu; v[1] = sv; u[2] = su; v[2] = ev; u[3] = eu; v[3] = ev; } else { if (has_clip) { x1 = std::max(x1, cmd.clip_x1); y1 = std::max(y1, cmd.clip_y1); x2 = std::min(x2, cmd.clip_x2); y2 = std::min(y2, cmd.clip_y2); } if (x2 <= x1 || y2 <= y1) continue; qx[0] = x1; qy[0] = y1; qx[1] = x2; qy[1] = y1; qx[2] = x1; qy[2] = y2; qx[3] = x2; qy[3] = y2; u[0] = su; v[0] = sv; u[1] = eu; v[1] = sv; u[2] = su; v[2] = ev; u[3] = eu; v[3] = ev; } const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string(); const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string(); const bool is_masked = !mask_name.empty(); // CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use // INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR. const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image ? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1)) : 1; const VkPipeline pipeline = get_pipeline(0, 0, blend_mode); const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures); for (int k = 0; k < 4; ++k) { Vertex& vert = vertices[v_count + k]; vert.position[0] = qx[k] * inv_w - 1.0f; vert.position[1] = 1.0f - qy[k] * inv_h; vert.position[2] = 0.0f; vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f; vert.uv[0] = u[k]; vert.uv[1] = v[k]; const auto& col = (k < 2) ? top_color : bot_color; vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3]; vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0; vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f; vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k]; vert.rhw = 1.0f; } indices[i_count + 0] = v_count + 0; indices[i_count + 1] = v_count + 1; indices[i_count + 2] = v_count + 2; indices[i_count + 3] = v_count + 2; indices[i_count + 4] = v_count + 1; indices[i_count + 3 + 2] = v_count + 3; const float mask_flag = is_masked ? -1.0f : 0.0f; if (!batches.empty() && batches.back().behind_3d == cmd.behind_3d && batches.back().pipeline == pipeline && batches.back().descriptor == descriptor && batches.back().constants.light_dir[3] == mask_flag) { batches.back().index_count += 6; } else { UiBatch batch{}; batch.first_index = i_count; batch.index_count = 6; batch.pipeline = pipeline; batch.descriptor = descriptor; batch.behind_3d = cmd.behind_3d; batch.constants.mvp = kIdentityMatrix; batch.constants.tint_color = {1.0f, 1.0f, 1.0f, 1.0f}; batch.constants.ambient_emissive = {1.0f, 1.0f, 1.0f, -1.0f}; batch.constants.light_dir = {0.0f, 0.0f, 1.0f, mask_flag}; batch.constants.light_diffuse = {0.0f, 0.0f, 0.0f, 0.0f}; batches.push_back(batch); } v_count += 4; i_count += 6; ++out_quad_count; } } std::uint32_t find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const { VkPhysicalDeviceMemoryProperties properties{}; vkGetPhysicalDeviceMemoryProperties(physical_, &properties); for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i) if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags) return i; throw std::runtime_error("no matching Vulkan memory type"); } void select_device() { std::uint32_t count = 0; check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count"); if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK"); std::vector candidates(count); check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices"); for (auto candidate : candidates) { std::uint32_t family_count = 0; vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr); std::vector families(family_count); vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data()); for (std::uint32_t i = 0; i < family_count; ++i) { VkBool32 present = VK_FALSE; check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support"); if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) { physical_ = candidate; queue_family_ = i; break; } } if (physical_) break; } if (!physical_) throw std::runtime_error("no graphics/present queue family"); VkPhysicalDeviceProperties properties{}; vkGetPhysicalDeviceProperties(physical_, &properties); device_name_ = properties.deviceName; timestamp_period_ns_ = properties.limits.timestampPeriod; VkPhysicalDeviceFeatures supported_features{}; vkGetPhysicalDeviceFeatures(physical_, &supported_features); VkPhysicalDeviceFeatures enabled_features{}; if (supported_features.samplerAnisotropy) { enabled_features.samplerAnisotropy = VK_TRUE; max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy); } std::uint32_t extension_count = 0; check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count"); std::vector available(extension_count); check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions"); std::vector extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME}; constexpr const char* portability_subset = "VK_KHR_portability_subset"; for (const auto& extension : available) if (std::strcmp(extension.extensionName, portability_subset) == 0) extensions.push_back(portability_subset); const float priority = 1.0f; VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority; VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info; device_info.enabledExtensionCount = static_cast(extensions.size()); device_info.ppEnabledExtensionNames = extensions.data(); device_info.pEnabledFeatures = &enabled_features; check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice"); vkGetDeviceQueue(device_, queue_family_, 0, &queue_); } void create_swapchain() { VkSurfaceCapabilitiesKHR capabilities{}; check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities"); std::uint32_t count = 0; check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count"); if (!count) throw std::runtime_error("surface has no formats"); std::vector formats(count); check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats"); VkSurfaceFormatKHR format = formats.front(); for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate; swapchain_format_ = format.format; int width = 0, height = 0; SDL_GetWindowSizeInPixels(window_, &width, &height); if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent; else { extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width); extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height); } present_mode_ = VK_PRESENT_MODE_FIFO_KHR; if (!vsync_) { std::uint32_t mode_count = 0; check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count"); std::vector modes(mode_count); if (mode_count) check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes"); if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end()) present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR; else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end()) present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR; } VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR}; info.surface = surface_; info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX); info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_; info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT; info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; info.preTransform = capabilities.currentTransform; info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_; info.clipped = VK_TRUE; check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR"); check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count"); std::vector images(count); check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images"); for (auto image : images) { VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; VkImageView handle = VK_NULL_HANDLE; check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView"); image_views_.push_back(handle); } VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; depth_info.imageType = VK_IMAGE_TYPE_2D; depth_info.format = VK_FORMAT_D32_SFLOAT; depth_info.extent = {extent_.width, extent_.height, 1}; depth_info.mipLevels = 1; depth_info.arrayLayers = 1; depth_info.samples = VK_SAMPLE_COUNT_1_BIT; depth_info.tiling = VK_IMAGE_TILING_OPTIMAL; depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT; depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth"); VkMemoryRequirements depth_reqs{}; vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs); VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; depth_alloc.allocationSize = depth_reqs.size; depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth"); check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth"); VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D; depth_view.format = VK_FORMAT_D32_SFLOAT; depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1; check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth"); } void create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) { VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; info.size = size; info.usage = usage; info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host"); VkMemoryRequirements reqs{}; vkGetBufferMemoryRequirements(device_, buffer, &reqs); VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; alloc.allocationSize = reqs.size; alloc.memoryTypeIndex = find_memory_type( reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host"); check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host"); check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host"); } void create_descriptors_and_buffers() { VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; sampler_info.magFilter = VK_FILTER_LINEAR; sampler_info.minFilter = VK_FILTER_LINEAR; sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT; sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT; sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; sampler_info.minLod = 0.0f; sampler_info.maxLod = VK_LOD_CLAMP_NONE; if (max_anisotropy_ > 1.0f) { sampler_info.anisotropyEnable = VK_TRUE; sampler_info.maxAnisotropy = max_anisotropy_; } check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler"); VkSamplerCreateInfo clamp_info = sampler_info; clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp"); VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; ui_sampler_info.magFilter = VK_FILTER_LINEAR; ui_sampler_info.minFilter = VK_FILTER_LINEAR; ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST; ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; ui_sampler_info.minLod = 0.0f; ui_sampler_info.maxLod = 0.0f; ui_sampler_info.anisotropyEnable = VK_FALSE; ui_sampler_info.maxAnisotropy = 1.0f; check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui"); VkDescriptorSetLayoutBinding tex_bindings[2]{}; tex_bindings[0].binding = 0; tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; tex_bindings[0].descriptorCount = 1; tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; tex_bindings[1].binding = 1; tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; tex_bindings[1].descriptorCount = 1; tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; layout_info.bindingCount = 2; layout_info.pBindings = tex_bindings; check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex"); VkDescriptorSetLayoutBinding bone_bindings[2]{}; bone_bindings[0].binding = 0; bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; bone_bindings[0].descriptorCount = 1; bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT; bone_bindings[1].binding = 1; bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; bone_bindings[1].descriptorCount = 1; bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; bone_layout_info.bindingCount = 2; bone_layout_info.pBindings = bone_bindings; check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone"); VkDescriptorPoolSize pool_sizes[3] = { {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4}, {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4}}; VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; pool_info.maxSets = 8192; pool_info.poolSizeCount = 3; pool_info.pPoolSizes = pool_sizes; check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool"); const std::uint8_t white_pixel[4] = {255, 255, 255, 255}; upload_texture(1, 1, white_pixel, fallback_texture_, false, false); fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); void* bone_raw = nullptr; create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, bone_buffer_, bone_memory_, &bone_raw); bone_mapped_ = static_cast(bone_raw); bone_capacity_ = kMaxBonesPerFrame; for (std::size_t i = 0; i < 16; ++i) bone_mapped_[i] = kIdentityMatrix[i]; VkPhysicalDeviceProperties properties{}; vkGetPhysicalDeviceProperties(physical_, &properties); const auto alignment = std::max( 1, properties.limits.minStorageBufferOffsetAlignment); state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; state_capacity_ = 256; void* state_raw = nullptr; create_host_buffer(state_capacity_ * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, state_buffer_, state_memory_, &state_raw); state_mapped_ = static_cast(state_raw); VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; bone_alloc.descriptorPool = descriptor_pool_; bone_alloc.descriptorSetCount = 1; bone_alloc.pSetLayouts = &bone_descriptor_layout_; check(vkAllocateDescriptorSets(device_, &bone_alloc, &bone_descriptor_set_), "vkAllocateDescriptorSets bone"); VkDescriptorBufferInfo buffer_infos[2] = { {bone_buffer_, 0, kBoneBufferBytes}, {state_buffer_, 0, sizeof(FixedFunctionState)}}; VkWriteDescriptorSet writes[2]{}; for (int binding = 0; binding < 2; ++binding) { writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; writes[binding].dstSet = bone_descriptor_set_; writes[binding].dstBinding = static_cast(binding); writes[binding].descriptorCount = 1; writes[binding].descriptorType = binding == 0 ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; writes[binding].pBufferInfo = &buffer_infos[binding]; } vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); void* ui_raw = nullptr; create_host_buffer( kUiBufferBytes, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, ui_buffer_, ui_memory_, &ui_raw); ui_mapped_ = static_cast(ui_raw); ui_vertex_capacity_ = kMaxUiVerticesPerFrame; ui_index_capacity_ = kMaxUiIndicesPerFrame; ui_vertex_bytes_ = kUiVertexBytes; } void ensure_bone_capacity(std::size_t required) { if (required <= bone_capacity_) return; VkPhysicalDeviceProperties properties{}; vkGetPhysicalDeviceProperties(physical_, &properties); const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); if (required > max_bones || required > UINT32_MAX) throw std::runtime_error("bone palette exceeds device storage-buffer limit"); std::size_t next = std::min(max_bones, std::max(required, bone_capacity_ * 2)); VkBuffer buffer = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; void* mapped = nullptr; create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); vkUnmapMemory(device_, bone_memory_); vkDestroyBuffer(device_, bone_buffer_, nullptr); vkFreeMemory(device_, bone_memory_, nullptr); bone_buffer_ = buffer; bone_memory_ = memory; bone_mapped_ = static_cast(mapped); bone_capacity_ = next; VkDescriptorBufferInfo info{bone_buffer_, 0, next * 16 * sizeof(float)}; VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; write.dstSet = bone_descriptor_set_; write.dstBinding = 0; write.descriptorCount = 1; write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; write.pBufferInfo = &info; vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); } void ensure_state_capacity(std::size_t required) { if (required <= state_capacity_) return; const auto next = std::max(required, state_capacity_ * 2); if (next > UINT32_MAX / state_stride_) throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); VkBuffer buffer = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; void* mapped = nullptr; create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); vkUnmapMemory(device_, state_memory_); vkDestroyBuffer(device_, state_buffer_, nullptr); vkFreeMemory(device_, state_memory_, nullptr); state_buffer_ = buffer; state_memory_ = memory; state_mapped_ = static_cast(mapped); state_capacity_ = next; VkDescriptorBufferInfo info{state_buffer_, 0, sizeof(FixedFunctionState)}; VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; write.dstSet = bone_descriptor_set_; write.dstBinding = 1; write.descriptorCount = 1; write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; write.pBufferInfo = &info; vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); } void ensure_ui_capacity(std::size_t command_count) { if (command_count <= ui_vertex_capacity_ / 4 && command_count <= ui_index_capacity_ / 6) return; if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); const std::size_t quads = std::max(command_count, ui_vertex_capacity_ / 2); const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) throw std::runtime_error("UI geometry exceeds 128 MiB safety limit"); VkBuffer buffer = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; void* mapped = nullptr; create_host_buffer(vertex_bytes + index_bytes, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); vkUnmapMemory(device_, ui_memory_); vkDestroyBuffer(device_, ui_buffer_, nullptr); vkFreeMemory(device_, ui_memory_, nullptr); ui_buffer_ = buffer; ui_memory_ = memory; ui_mapped_ = static_cast(mapped); ui_vertex_capacity_ = quads * 4; ui_index_capacity_ = quads * 6; ui_vertex_bytes_ = vertex_bytes; } void create_render_pass_and_layout() { VkAttachmentDescription attachments[2]{}; attachments[0].format = swapchain_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[0].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; VkSubpassDescription subpass{}; subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; subpass.pDepthStencilAttachment = &depth_ref; VkSubpassDependency dependency{}; dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0; dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency; check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass"); for (auto view : image_views_) { VkImageView fb_attachments[2] = {view, depth_view_}; VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; framebuffer.renderPass = pass_; framebuffer.attachmentCount = 2; framebuffer.pAttachments = fb_attachments; framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; VkFramebuffer handle = VK_NULL_HANDLE; check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer"); framebuffers_.push_back(handle); } const auto vert = read_spirv("native.vert.spv"); const auto frag = read_spirv("native.frag.spv"); auto module = [&](const std::vector& code) { VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO}; info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data(); VkShaderModule handle = VK_NULL_HANDLE; check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule"); return handle; }; vertex_module_ = module(vert); fragment_module_ = module(frag); VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_}; VkPushConstantRange push_constants{}; push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; push_constants.size = sizeof(PushConstants); VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO}; layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts; layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants; check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); get_pipeline(0, 2, 0); } VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, bool lines = false, std::uint8_t z_func = 4) { const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | (std::uint32_t(z_func & 0xfu) << 25); if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; VkPipelineShaderStageCreateInfo stages[2]{}; stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main"; stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main"; VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX}; VkVertexInputAttributeDescription attributes[8] = { {0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)}, {1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)}, {2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)}, {3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)}, {4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)}, {5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)}, {6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)}, {7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)}}; VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO}; input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding; input.vertexAttributeDescriptionCount = 8; input.pVertexAttributeDescriptions = attributes; VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO}; assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; VkRect2D scissor{{0, 0}, extent_}; VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO}; raster.polygonMode = VK_POLYGON_MODE_FILL; raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE); raster.frontFace = VK_FRONT_FACE_CLOCKWISE; raster.lineWidth = 1; VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; multisample.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT; VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE; switch (z_func) { case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break; case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break; case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break; case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break; case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break; case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break; case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break; default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break; } auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor { switch (d3d) { case 1: return VK_BLEND_FACTOR_ZERO; case 2: return VK_BLEND_FACTOR_ONE; case 3: return VK_BLEND_FACTOR_SRC_COLOR; case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR; case 5: return VK_BLEND_FACTOR_SRC_ALPHA; case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; case 7: return VK_BLEND_FACTOR_DST_ALPHA; case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; case 9: return VK_BLEND_FACTOR_DST_COLOR; case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR; case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE; default: return fallback; } }; VkPipelineColorBlendAttachmentState blend_attachment{}; blend_attachment.colorWriteMask = 0xf; if (blend > 0) { blend_attachment.blendEnable = VK_TRUE; if (blend >= 16) { blend_attachment.srcColorBlendFactor = d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA); blend_attachment.dstColorBlendFactor = d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA); } else { blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA; blend_attachment.dstColorBlendFactor = blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; } blend_attachment.colorBlendOp = VK_BLEND_OP_ADD; blend_attachment.srcAlphaBlendFactor = VK_BLEND_FACTOR_ONE; blend_attachment.dstAlphaBlendFactor = VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD; } VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO}; blending.attachmentCount = 1; blending.pAttachments = &blend_attachment; VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO}; pipeline_info.stageCount = 2; pipeline_info.pStages = stages; pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly; pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster; pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; pipeline_info.pColorBlendState = &blending; pipeline_info.layout = layout_; pipeline_info.renderPass = pass_; VkPipeline handle = VK_NULL_HANDLE; check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); pipelines_.emplace(key, handle); return handle; } VkDescriptorSet allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, VkSampler sampler0 = VK_NULL_HANDLE, VkSampler sampler1 = VK_NULL_HANDLE) { VkDescriptorSet set = VK_NULL_HANDLE; VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; desc_alloc.descriptorPool = descriptor_pool_; desc_alloc.descriptorSetCount = 1; desc_alloc.pSetLayouts = &descriptor_layout_; check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets"); VkDescriptorImageInfo image_infos[2]{}; image_infos[0].sampler = sampler0 ? sampler0 : sampler_; image_infos[0].imageView = view0; image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_); image_infos[1].imageView = view1; image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; VkWriteDescriptorSet writes[2]{}; for (int b = 0; b < 2; ++b) { writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; writes[b].dstSet = set; writes[b].dstBinding = static_cast(b); writes[b].descriptorCount = 1; writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; writes[b].pImageInfo = &image_infos[b]; } vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); return set; } VkDescriptorSet allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) { VkDescriptorSet set = VK_NULL_HANDLE; VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; desc_alloc.descriptorPool = descriptor_pool_; desc_alloc.descriptorSetCount = 1; desc_alloc.pSetLayouts = &descriptor_layout_; check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui"); VkDescriptorImageInfo image_infos[2]{}; image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_; image_infos[0].imageView = view0; image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_); image_infos[1].imageView = view1; image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; VkWriteDescriptorSet writes[2]{}; for (int b = 0; b < 2; ++b) { writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; writes[b].dstSet = set; writes[b].dstBinding = static_cast(b); writes[b].descriptorCount = 1; writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; writes[b].pImageInfo = &image_infos[b]; } vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); return set; } GpuTexture& get_gpu_texture( const std::string& texture_name, const std::unordered_map>& capture_textures) { if (texture_name.empty()) return fallback_texture_; auto [it, inserted] = textures_.try_emplace(texture_name); it->second.last_used_frame = frame_number_; if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) return it->second; if (inserted) { it->second.descriptor = fallback_texture_.descriptor; it->second.ui_descriptor = fallback_texture_.ui_descriptor; } it->second.last_decode_attempt_frame = frame_number_; mtgodot::Image decoded; if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) { decoded = decode_texture_bytes(raw->second.data(), raw->second.size()); } #ifdef MT_NATIVE_HAS_LIVE_CLIENT if (!decoded.ok()) { if (texture_name.rfind("mem:", 0) == 0) { UIMemoryTexture mem_tex; if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { const auto mtra = native_draw_capture::encode_raw_argb_as_mtra( static_cast(mem_tex.width), static_cast(mem_tex.height), mem_tex.argb.data()); decoded = decode_texture_bytes(mtra.data(), mtra.size()); } } else { std::vector pack_bytes; if (read_live_pack_texture(texture_name, pack_bytes)) decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size()); } } #endif if (!decoded.ok()) return it->second; bool is_ui_tex = false; if (texture_name.rfind("icon/", 0) == 0 || texture_name.rfind("d:/ymir work/ui/", 0) == 0 || texture_name.rfind("d:\\ymir work\\ui\\", 0) == 0 || texture_name.rfind("locale/", 0) == 0 || texture_name.rfind("locale\\", 0) == 0 || texture_name.rfind("mem:", 0) == 0 || (decoded.w <= 128 && decoded.h <= 128)) { is_ui_tex = true; } upload_texture(decoded.w, decoded.h, decoded.rgba.data(), it->second, true, !is_ui_tex); it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view); it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view); return it->second; } std::uint64_t sampler_state_key(const Render3DDraw& draw, int stage) const { return (std::uint64_t(draw.address_u[stage] & 15u)) | (std::uint64_t(draw.address_v[stage] & 15u) << 4) | (std::uint64_t(draw.min_filter[stage] & 15u) << 8) | (std::uint64_t(draw.mag_filter[stage] & 15u) << 12) | (std::uint64_t(draw.mip_filter[stage] & 15u) << 16); } VkSampler sampler_for_draw(const Render3DDraw& draw, int stage) { const auto key = sampler_state_key(draw, stage); if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_; if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second; auto address_mode = [](std::uint32_t mode) { switch (mode) { case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT; case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER; default: return VK_SAMPLER_ADDRESS_MODE_REPEAT; } }; VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; info.addressModeU = address_mode(draw.address_u[stage]); info.addressModeV = address_mode(draw.address_v[stage]); info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE; info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR; info.minLod = 0.0f; info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE; if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) { info.anisotropyEnable = VK_TRUE; info.maxAnisotropy = max_anisotropy_; } VkSampler sampler = VK_NULL_HANDLE; check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state"); state_samplers_.emplace(key, sampler); return sampler; } VkDescriptorSet get_texture_descriptor( const std::string& texture0_name, const std::string& mask_name, const std::unordered_map>& capture_textures, const Render3DDraw& draw) { const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures); const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" + std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1)); if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { it->second.last_used_frame = frame_number_; return it->second.set; } const VkDescriptorSet set = allocate_texture_descriptor_set( tex0.view ? tex0.view : fallback_texture_.view, tex1.view ? tex1.view : fallback_texture_.view, sampler_for_draw(draw, 0), sampler_for_draw(draw, 1)); paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); return set; } VkDescriptorSet get_ui_texture_descriptor( const std::string& texture0_name, const std::string& mask_name, const std::unordered_map>& capture_textures) { const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; const auto& tex1 = get_gpu_texture(mask_name, capture_textures); if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name; if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { it->second.last_used_frame = frame_number_; return it->second.set; } const VkDescriptorSet set = allocate_ui_texture_descriptor_set( tex0.view ? tex0.view : fallback_texture_.view, tex1.view ? tex1.view : fallback_texture_.view); paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); return set; } void upload_texture( std::uint32_t width, std::uint32_t height, const std::uint8_t* rgba, GpuTexture& texture, bool count_upload, bool generate_mips = true) { const VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; const std::uint32_t mip_levels = generate_mips ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) : 1u; VkBuffer staging_buffer = VK_NULL_HANDLE; VkDeviceMemory staging_memory = VK_NULL_HANDLE; VkBufferCreateInfo buffer_info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; buffer_info.size = byte_size; buffer_info.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT; buffer_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; check(vkCreateBuffer(device_, &buffer_info, nullptr, &staging_buffer), "vkCreateBuffer staging"); VkMemoryRequirements buffer_reqs{}; vkGetBufferMemoryRequirements(device_, staging_buffer, &buffer_reqs); VkMemoryAllocateInfo buffer_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; buffer_alloc.allocationSize = buffer_reqs.size; buffer_alloc.memoryTypeIndex = find_memory_type( buffer_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); check(vkAllocateMemory(device_, &buffer_alloc, nullptr, &staging_memory), "vkAllocateMemory staging"); check(vkBindBufferMemory(device_, staging_buffer, staging_memory, 0), "vkBindBufferMemory staging"); void* mapped = nullptr; check(vkMapMemory(device_, staging_memory, 0, byte_size, 0, &mapped), "vkMapMemory staging"); std::memcpy(mapped, rgba, byte_size); vkUnmapMemory(device_, staging_memory); VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; image_info.imageType = VK_IMAGE_TYPE_2D; image_info.format = VK_FORMAT_R8G8B8A8_UNORM; image_info.extent = {width, height, 1}; image_info.mipLevels = mip_levels; image_info.arrayLayers = 1; image_info.samples = VK_SAMPLE_COUNT_1_BIT; image_info.tiling = VK_IMAGE_TILING_OPTIMAL; image_info.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT; image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture"); VkMemoryRequirements image_reqs{}; vkGetImageMemoryRequirements(device_, texture.image, &image_reqs); VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; image_alloc.allocationSize = image_reqs.size; image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); VkCommandBuffer upload_cmd = VK_NULL_HANDLE; VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; to_transfer.image = texture.image; to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; vkCmdPipelineBarrier( upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &to_transfer); VkBufferImageCopy region{}; region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; region.imageExtent = {width, height, 1}; vkCmdCopyBufferToImage(upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ®ion); std::int32_t mip_w = static_cast(width); std::int32_t mip_h = static_cast(height); for (std::uint32_t i = 1; i < mip_levels; ++i) { VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; barrier.image = texture.image; barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1}; barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; vkCmdPipelineBarrier( upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); const std::int32_t next_w = std::max(1, mip_w / 2); const std::int32_t next_h = std::max(1, mip_h / 2); VkImageBlit blit{}; blit.srcOffsets[0] = {0, 0, 0}; blit.srcOffsets[1] = {mip_w, mip_h, 1}; blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1}; blit.dstOffsets[0] = {0, 0, 0}; blit.dstOffsets[1] = {next_w, next_h, 1}; blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; vkCmdBlitImage( upload_cmd, texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &blit, VK_FILTER_LINEAR); barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; vkCmdPipelineBarrier( upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier); mip_w = next_w; mip_h = next_h; } VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; to_shader.image = texture.image; to_shader.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1}; vkCmdPipelineBarrier( upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, 0, 0, nullptr, 0, nullptr, 1, &to_shader); check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); vkDestroyBuffer(device_, staging_buffer, nullptr); vkFreeMemory(device_, staging_memory, nullptr); VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = VK_FORMAT_R8G8B8A8_UNORM; view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture"); if (count_upload) { ++texture_upload_count_; texture_uploaded_bytes_ += byte_size; } } void release_texture(GpuTexture& texture) { for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor}) if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor) vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set); if (texture.view) vkDestroyImageView(device_, texture.view, nullptr); if (texture.image) vkDestroyImage(device_, texture.image, nullptr); if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr); texture = {}; } void prune_textures() { constexpr std::uint64_t static_retention_frames = 120; constexpr std::uint64_t memory_retention_frames = 2; for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) { const auto retention = it->first.find("mem:") != std::string::npos ? memory_retention_frames : static_retention_frames; if (frame_number_ - it->second.last_used_frame > retention) { vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set); it = paired_descriptors_.erase(it); } else ++it; } for (auto it = textures_.begin(); it != textures_.end();) { const auto retention = it->first.rfind("mem:", 0) == 0 ? memory_retention_frames : static_retention_frames; if (frame_number_ - it->second.last_used_frame > retention) { release_texture(it->second); it = textures_.erase(it); } else ++it; } } void release_geometry(Geometry& geometry) { if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr); if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr); geometry = {}; } void upload_geometry(const Render3DDraw& draw, Geometry& geometry) { if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) throw std::runtime_error("invalid native geometry array size"); const auto count = draw.positions.size() / 3; if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX) throw std::runtime_error("geometry exceeds Vulkan index range"); const std::uint32_t default_argb = (!draw.texture0.empty() || draw.lighting != 0) ? 0xffffffffu : 0xff4fbfffu; const bool has_skin = draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4; std::vector vertices(count); for (std::size_t i = 0; i < count; ++i) { vertices[i].position[0] = draw.positions[3*i]; vertices[i].position[1] = draw.positions[3*i+1]; vertices[i].position[2] = draw.positions[3*i+2]; vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f; if (3*i + 2 < draw.normals.size()) { vertices[i].normal[0] = draw.normals[3*i]; vertices[i].normal[1] = draw.normals[3*i+1]; vertices[i].normal[2] = draw.normals[3*i+2]; } else { vertices[i].normal[0] = 0.0f; vertices[i].normal[1] = 0.0f; vertices[i].normal[2] = 1.0f; } if (2*i + 1 < draw.uv0.size()) { vertices[i].uv[0] = draw.uv0[2*i]; vertices[i].uv[1] = draw.uv0[2*i+1]; } else { vertices[i].uv[0] = 0.0f; vertices[i].uv[1] = 0.0f; } const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb; vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f; vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f; vertices[i].color[2] = float(argb & 255) / 255.0f; vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f; if (has_skin) { for (int k = 0; k < 4; ++k) { vertices[i].joints[k] = draw.bone_indices[4*i + k]; vertices[i].weights[k] = draw.bone_weights[4*i + k]; } } else { vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0; vertices[i].weights[0] = 1.0f; vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f; } if (2*i + 1 < draw.uv1.size()) { vertices[i].mask_uv[0] = draw.uv1[2*i]; vertices[i].mask_uv[1] = draw.uv1[2*i+1]; } else { vertices[i].mask_uv[0] = 0.0f; vertices[i].mask_uv[1] = 0.0f; } } for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count"); geometry.index_offset = vertices.size() * sizeof(Vertex); const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t); if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB"); VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT; info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer"); VkMemoryRequirements requirements{}; vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements); VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; allocation.allocationSize = requirements.size; allocation.memoryTypeIndex = find_memory_type( requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory"); check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory"); void* mapped = nullptr; check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory"); std::memcpy(mapped, vertices.data(), geometry.index_offset); std::memcpy(static_cast(mapped) + geometry.index_offset, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); vkUnmapMemory(device_, geometry.memory); geometry.vertex_count = static_cast(count); geometry.index_count = static_cast(draw.indices.size()); ++upload_count_; uploaded_bytes_ += bytes; } bool vsync_ = true; VkPresentModeKHR present_mode_ = VK_PRESENT_MODE_FIFO_KHR; float timestamp_period_ns_ = 1.0f; float max_anisotropy_ = 1.0f; SDL_Window* window_ = nullptr; VkInstance instance_ = VK_NULL_HANDLE; VkSurfaceKHR surface_ = VK_NULL_HANDLE; VkPhysicalDevice physical_ = VK_NULL_HANDLE; VkDevice device_ = VK_NULL_HANDLE; VkQueue queue_ = VK_NULL_HANDLE; std::uint32_t queue_family_ = 0; std::string device_name_; VkSwapchainKHR swapchain_ = VK_NULL_HANDLE; VkFormat swapchain_format_ = VK_FORMAT_UNDEFINED; VkExtent2D extent_{}; std::vector image_views_; VkImage depth_image_ = VK_NULL_HANDLE; VkDeviceMemory depth_memory_ = VK_NULL_HANDLE; VkImageView depth_view_ = VK_NULL_HANDLE; VkRenderPass pass_ = VK_NULL_HANDLE; std::vector framebuffers_; VkSampler sampler_ = VK_NULL_HANDLE; VkSampler clamp_sampler_ = VK_NULL_HANDLE; VkSampler ui_sampler_ = VK_NULL_HANDLE; std::unordered_map state_samplers_; VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE; VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE; VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE; VkDescriptorSet bone_descriptor_set_ = VK_NULL_HANDLE; VkBuffer bone_buffer_ = VK_NULL_HANDLE; VkDeviceMemory bone_memory_ = VK_NULL_HANDLE; float* bone_mapped_ = nullptr; std::size_t bone_capacity_ = 0; VkBuffer state_buffer_ = VK_NULL_HANDLE; VkDeviceMemory state_memory_ = VK_NULL_HANDLE; std::uint8_t* state_mapped_ = nullptr; VkDeviceSize state_stride_ = 0; std::size_t state_capacity_ = 0; VkBuffer ui_buffer_ = VK_NULL_HANDLE; VkDeviceMemory ui_memory_ = VK_NULL_HANDLE; std::uint8_t* ui_mapped_ = nullptr; std::size_t ui_vertex_capacity_ = 0, ui_index_capacity_ = 0; VkDeviceSize ui_vertex_bytes_ = 0; GpuTexture fallback_texture_{}; std::unordered_map textures_; std::unordered_map paired_descriptors_; VkShaderModule vertex_module_ = VK_NULL_HANDLE; VkShaderModule fragment_module_ = VK_NULL_HANDLE; VkPipelineLayout layout_ = VK_NULL_HANDLE; std::unordered_map pipelines_; std::unordered_map geometries_; std::uint64_t frame_number_ = 0; std::uint64_t timed_frames_ = 0; std::size_t upload_count_ = 0, uploaded_bytes_ = 0; std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0; VkCommandPool pool_ = VK_NULL_HANDLE; VkCommandBuffer command_ = VK_NULL_HANDLE; VkQueryPool query_pool_ = VK_NULL_HANDLE; bool has_pending_query_ = false; bool os_cursor_hidden_ = false; bool hardware_cursor_enabled_ = false; std::array game_cursors_{}; SDL_Cursor* fallback_cursor_ = nullptr; int current_cursor_shape_ = -1; int last_mx_ = 0, last_my_ = 0; VkSemaphore acquire_ = VK_NULL_HANDLE, rendered_ = VK_NULL_HANDLE; VkFence fence_ = VK_NULL_HANDLE; std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0; std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0; std::size_t last_vertex_count_ = 0, last_index_count_ = 0; Timings timings_{}; }; std::vector make_test_draws(std::size_t count, std::size_t triangles_per_draw) { std::vector draws; draws.reserve(count); for (std::size_t i = 0; i < count; ++i) { Render3DDraw draw{}; draw.geometry_key = i + 1; draw.geometry_revision = 1; draw.z_enable = 1; draw.z_write = 1; for (auto* matrix : {draw.world, draw.view, draw.proj}) for (int diagonal = 0; diagonal < 4; ++diagonal) matrix[diagonal * 5] = 1.0f; const float x = (float(i % 8) - 3.5f) * 0.24f; const float y = (float(i / 8) - 3.5f) * 0.22f; draw.positions.reserve(triangles_per_draw * 9); draw.indices.reserve(triangles_per_draw * 3); draw.diffuse.reserve(triangles_per_draw * 3); for (std::size_t t = 0; t < triangles_per_draw; ++t) { const float tx = x + (float(t % 16) - 7.5f) * 0.009f; const float ty = y + (float(t / 16) - float(triangles_per_draw / 32)) * 0.006f; const auto base = static_cast(draw.positions.size() / 3); draw.positions.insert(draw.positions.end(), {tx-0.004f, ty-0.004f, 0.5f, tx+0.004f, ty-0.004f, 0.5f, tx, ty+0.004f, 0.5f}); draw.indices.insert(draw.indices.end(), {base, base+1, base+2}); draw.diffuse.insert(draw.diffuse.end(), {0xff4fbfff, 0xff4fbfff, 0xff4fbfff}); } draws.push_back(std::move(draw)); } return draws; } void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms = 0.0, std::vector frame_ms = {}) { const auto timing = renderer.timings(); std::sort(frame_ms.begin(), frame_ms.end()); const auto percentile = [&](double fraction) { if (frame_ms.empty()) return 0.0; const auto index = static_cast(std::ceil(fraction * frame_ms.size())) - 1; return frame_ms[std::min(index, frame_ms.size() - 1)]; }; std::cout << "device=" << renderer.device_name() << " present_mode=" << renderer.present_mode_name() << " frames=" << completed << " draws=" << renderer.draw_count() << " skinned_draws=" << renderer.skinned_draw_count() << " ui_batches=" << renderer.ui_batch_count() << " ui_quads=" << renderer.ui_quad_count() << " vertices=" << renderer.vertex_count() << " indices=" << renderer.index_count() << " uploads=" << renderer.upload_count() << " uploaded_bytes=" << renderer.uploaded_bytes() << " texture_uploads=" << renderer.texture_upload_count() << " texture_bytes=" << renderer.texture_uploaded_bytes() << " mean_frame_ms=" << (completed ? wall_ms / completed : 0.0) << " p95_frame_ms=" << percentile(0.95) << " p99_frame_ms=" << percentile(0.99) << " max_frame_ms=" << percentile(1.0) << " game_update_ms=" << (completed ? update_ms / completed : 0.0) << " sync_ms=" << (completed ? timing.sync_ms / completed : 0.0) << " prepare_ms=" << (completed ? timing.prepare_ms / completed : 0.0) << " steady_prepare_ms=" << (completed > 1 ? timing.steady_prepare_ms / (completed - 1) : 0.0) << " submit_ms=" << (completed ? timing.submit_ms / completed : 0.0) << " present_ms=" << (completed ? timing.present_ms / completed : 0.0) << " gpu_ms=" << (timing.gpu_samples ? timing.gpu_ms / double(timing.gpu_samples) : 0.0) << '\n'; } #ifdef MT_NATIVE_HAS_LIVE_CLIENT bool py_exec(const std::string& code) { std::string err; if (!PythonBoot::RunLine(code.c_str(), &err)) { std::cerr << "PythonBoot::RunLine failed: " << err << '\n'; return false; } return true; } bool py_eval_true(const char* expr) { std::string res, err; return PythonBoot::Evaluate(expr, &res, &err) && (res == "True" || res == "1"); } bool parse_live_server_spec(const std::string& spec, std::string& host, int& auth_port, int& game_port) { const auto p1 = spec.find(':'); if (p1 == std::string::npos) return false; const auto p2 = spec.find(':', p1 + 1); if (p2 == std::string::npos) return false; host = spec.substr(0, p1); auth_port = std::stoi(spec.substr(p1 + 1, p2 - p1 - 1)); game_port = std::stoi(spec.substr(p2 + 1)); return !host.empty() && auth_port > 0 && game_port > 0; } int run_live_client( VulkanWindow& renderer, const std::string& client_dir, int frames, int fake_mobs, bool gpu_skinning, bool native_terrain, bool login_screen, const std::string& live_server_spec, const std::string& capture_out) { if (fake_mobs > 0) { const std::string mobs_str = std::to_string(fake_mobs); setenv("MT_FAKE_MOB_COUNT", mobs_str.c_str(), 1); } char resolved_client_dir[4096] = {}; const std::string abs_client_dir = realpath(client_dir.c_str(), resolved_client_dir) ? std::string(resolved_client_dir) : client_dir; setenv("MT_40250_CLIENT", abs_client_dir.c_str(), 1); if (chdir(abs_client_dir.c_str()) != 0 || !mtpack40250::initialize(".")) { throw std::runtime_error("cannot initialize 40250 pack at: " + abs_client_dir); } const bool host_hardware_cursor = !renderer.touch_controller.is_enabled(); PythonBoot::SetHostHardwareCursorEnabled(host_hardware_cursor); if (host_hardware_cursor) renderer.enable_game_hardware_cursor(); SetNativeTerrainRenderEnabled(native_terrain); SetGpuSkinningEnabled(gpu_skinning); NativeAudioEngine audio; std::string server_host = "127.0.0.1"; int auth_port = 0; int game_port = 0; const bool use_external_server = !live_server_spec.empty(); FakeLoginServer server; if (use_external_server) { if (!parse_live_server_spec(live_server_spec, server_host, auth_port, game_port)) throw std::runtime_error("invalid --live-server spec (expected HOST:AUTH_PORT:GAME_PORT): " + live_server_spec); } else { if (!server.Start()) throw std::runtime_error("FakeLoginServer failed to start on loopback"); auth_port = server.AuthPort(); game_port = server.GamePort(); } std::string error; const char* env_stdlib = std::getenv("MT_PYTHON_STDLIB"); std::string stdlib = (env_stdlib && *env_stdlib) ? env_stdlib : bundled_file_path("python27.zip"); #ifdef __ANDROID__ if (!env_stdlib || !*env_stdlib) { std::size_t zip_size = 0; void* zip = SDL_LoadFile("assets://python27.zip", &zip_size); if (!zip) zip = SDL_LoadFile("python27.zip", &zip_size); if (!zip) throw std::runtime_error("cannot load bundled python27.zip"); char* pref = SDL_GetPrefPath("mtgodot", "native-render"); if (!pref) { SDL_free(zip); throw std::runtime_error("cannot locate app data directory"); } stdlib = std::string(pref) + "python27.zip"; SDL_free(pref); std::ofstream output(stdlib, std::ios::binary | std::ios::trunc); output.write(static_cast(zip), static_cast(zip_size)); SDL_free(zip); if (!output) throw std::runtime_error("cannot extract bundled python27.zip"); } #endif if (!PythonBoot::Start(stdlib.c_str(), &error)) throw std::runtime_error("PythonBoot::Start: " + error); SetPlatformServerTime(123456789); PythonBoot::SetUISize(static_cast(renderer.logical_width()), static_cast(renderer.logical_height())); if (!PythonBoot::RunMainScript("", &error) || !PythonBoot::IsAppLooping()) throw std::runtime_error("PythonBoot::RunMainScript: " + error); const std::unordered_map> empty_textures; auto pump_until = [&](double max_seconds, int sleep_ms, const auto& cond) -> bool { const auto deadline = std::chrono::steady_clock::now() + std::chrono::duration(max_seconds); while (std::chrono::steady_clock::now() < deadline && renderer.poll(true) && PythonBoot::IsAppLooping()) { PythonBoot::UIUpdate(); renderer.sync_game_cursor(); PythonBoot::UIRender(); audio.pump(); unsigned ui_w = renderer.width(), ui_h = renderer.height(); UIRenderGetSize(&ui_w, &ui_h); renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); if (cond()) return true; if (sleep_ms > 0) std::this_thread::sleep_for(std::chrono::milliseconds(sleep_ms)); } return cond(); }; if (!py_exec("import __main__, app, networkModule, introLogin, introSelect, introLoading, game\n" "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]")) throw std::runtime_error("failed to locate networkModule.MainStream"); if (!pump_until(20.0, 5, [] { return py_eval_true("isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)"); })) throw std::runtime_error("timed out waiting for LoginWindow"); if (login_screen) { char conn_cmd[512]; std::snprintf( conn_cmd, sizeof(conn_cmd), "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" "_w = _stream.curPhaseWindow\n" "_w._LoginWindow__OpenServerBoard()", server_host.c_str(), game_port, server_host.c_str(), auth_port); if (!py_exec(conn_cmd)) throw std::runtime_error("failed to configure LoginWindow connection info"); for (int i = 0; i < 15; ++i) { PythonBoot::UIUpdate(); renderer.sync_game_cursor(); PythonBoot::UIRender(); } } else { char login_cmd[512]; std::snprintf( login_cmd, sizeof(login_cmd), "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" "_w = _stream.curPhaseWindow\n" "_w._LoginWindow__OpenLoginBoard()\n" "_w.idEditLine.SetText('%s')\n" "_w.pwdEditLine.SetText('%s')\n" "_w._LoginWindow__OnClickLoginButton()", server_host.c_str(), game_port, server_host.c_str(), auth_port, FakeLoginServer::kLogin, FakeLoginServer::kPassword); if (!py_exec(login_cmd)) throw std::runtime_error("failed to submit login credentials"); if (!pump_until(20.0, 5, [&] { return (use_external_server || server.Has("game1:login2")) && py_eval_true("isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)"); })) throw std::runtime_error("timed out waiting for SelectCharacterWindow"); if (!py_exec("_stream.curPhaseWindow.SelectSlot(0)\n" "_stream.curPhaseWindow.StartGame()")) throw std::runtime_error("failed to start game from SelectCharacterWindow"); if (!use_external_server) { if (!pump_until(30.0, 5, [&] { return server.Has("game2:client_version"); })) throw std::runtime_error("timed out waiting for LoadingWindow (client_version)"); } if (!pump_until(60.0, 0, [&] { return (use_external_server || server.Has("game2:burst_pong")) && py_eval_true("isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()"); })) throw std::runtime_error("timed out waiting for GameWindow (server error: " + server.Error() + ")"); const int warmup_frames = std::max(10, fake_mobs / 2 + 10); for (int i = 0; i < warmup_frames; ++i) { if (!renderer.poll(true) || !PythonBoot::IsAppLooping()) break; PythonBoot::UIUpdate(); renderer.sync_game_cursor(); audio.pump(); } } // Render one warmup frame to upload initial scene & UI textures/geometries before timed benchmark. { unsigned ui_w = renderer.width(), ui_h = renderer.height(); UIRenderGetSize(&ui_w, &ui_h); renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); } if (!capture_out.empty()) { std::unordered_map> cap_textures; auto add_tex = [&](const std::string& name) { if (name.empty() || cap_textures.count(name)) return; if (name.rfind("mem:", 0) == 0) { UIMemoryTexture mem_tex; if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { auto mtra = native_draw_capture::encode_raw_argb_as_mtra( static_cast(mem_tex.width), static_cast(mem_tex.height), mem_tex.argb.data()); if (!mtra.empty()) cap_textures.emplace(name, std::move(mtra)); } return; } std::vector bytes; if (read_live_pack_texture(name, bytes)) cap_textures.emplace(name, std::move(bytes)); }; for (const auto& d : Render3DDraws()) { add_tex(d.texture0); add_tex(d.texture1); } for (const auto& c : UIRenderCommands()) { if (c.kind == UIRenderCommand::Image) { add_tex(c.text); add_tex(c.mask); } } unsigned ui_w = renderer.width(), ui_h = renderer.height(); UIRenderGetSize(&ui_w, &ui_h); std::vector cap_ui = UIRenderCommands(); if (renderer.touch_controller.is_enabled()) { renderer.touch_controller.update_screen_size(int(ui_w), int(ui_h)); renderer.touch_controller.append_ui_commands(cap_ui); } native_draw_capture::write(capture_out, Render3DDraws(), cap_textures, ui_w, ui_h, cap_ui); } renderer.reset_timings(); const auto start = std::chrono::steady_clock::now(); double update_ms = 0.0; int completed = 0; std::vector frame_ms; if (frames > 0) frame_ms.reserve(static_cast(frames)); while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) { const auto frame_start = std::chrono::steady_clock::now(); const auto t0 = std::chrono::steady_clock::now(); PythonBoot::UIUpdate(); renderer.sync_game_cursor(); PythonBoot::UIRender(); audio.pump(); const auto t1 = std::chrono::steady_clock::now(); update_ms += std::chrono::duration(t1 - t0).count(); unsigned ui_w = renderer.width(), ui_h = renderer.height(); UIRenderGetSize(&ui_w, &ui_h); renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); frame_ms.push_back(std::chrono::duration( std::chrono::steady_clock::now() - frame_start).count()); ++completed; } renderer.finish_gpu_timings(); const auto wall_ms = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); print_summary(renderer, completed, wall_ms, update_ms, std::move(frame_ms)); PythonBoot::Stop(); if (!use_external_server) server.Stop(); return (frames == 0 || completed == frames) ? 0 : 2; } #endif } // namespace int main(int argc, char** argv) { try { int frames = 180; bool frames_specified = false; bool synthetic_specified = false; bool interactive = false; bool login_screen = false; int init_width = 1280; int init_height = 800; std::size_t draw_count = 64; std::size_t triangles_per_draw = 333; std::string capture_path; std::string capture_out; std::string live_client_dir; std::string live_server_spec; int fake_mobs = 64; bool animate_first_draw = false; bool animate_bones = false; bool vsync = true; bool gpu_skinning = true; bool native_terrain = true; bool mobile_mode = false; #ifdef __ANDROID__ mobile_mode = true; #endif for (int i = 1; i < argc; ++i) { const std::string arg = argv[i]; if (arg == "--frames" && i+1 < argc) { frames = std::stoi(argv[++i]); frames_specified = true; } else if (arg == "--interactive") interactive = true; else if (arg == "--login-screen") { login_screen = true; interactive = true; } else if (arg == "--live-server" && i+1 < argc) { live_server_spec = argv[++i]; interactive = true; } else if (arg == "--width" && i+1 < argc) init_width = std::stoi(argv[++i]); else if (arg == "--height" && i+1 < argc) init_height = std::stoi(argv[++i]); else if (arg == "--draws" && i+1 < argc) { draw_count = std::stoul(argv[++i]); synthetic_specified = true; } else if (arg == "--triangles-per-draw" && i+1 < argc) { triangles_per_draw = std::stoul(argv[++i]); synthetic_specified = true; } else if (arg == "--capture" && i+1 < argc) { capture_path = argv[++i]; synthetic_specified = true; } else if (arg == "--capture-out" && i+1 < argc) capture_out = argv[++i]; else if (arg == "--live-client" && i+1 < argc) live_client_dir = argv[++i]; else if (arg == "--fake-mobs" && i+1 < argc) fake_mobs = std::stoi(argv[++i]); else if (arg == "--animate-first-draw") { animate_first_draw = true; synthetic_specified = true; } else if (arg == "--animate-bones") { animate_bones = true; synthetic_specified = true; } else if (arg == "--no-vsync") vsync = false; else if (arg == "--gpu-skinning") gpu_skinning = true; else if (arg == "--no-gpu-skinning") gpu_skinning = false; else if (arg == "--no-terrain") native_terrain = false; else if (arg == "--mobile") mobile_mode = true; else throw std::runtime_error( "usage: mt_native_render [--frames N] [--interactive] [--login-screen] " "[--live-server HOST:AUTH_PORT:GAME_PORT] [--width W] [--height H] " "[--draws N] [--triangles-per-draw N] [--capture FILE] [--animate-first-draw] " "[--animate-bones] [--no-vsync] [--live-client DIR] [--fake-mobs N] " "[--gpu-skinning|--no-gpu-skinning] [--no-terrain] [--mobile] [--capture-out FILE]"); } if (live_client_dir.empty() && !synthetic_specified) { const char* env_client = std::getenv("MT_40250_CLIENT"); if (env_client && *env_client && access((std::string(env_client) + "/pack/Index").c_str(), R_OK) == 0) { live_client_dir = env_client; } else if (access("Client/pack/Index", R_OK) == 0) { live_client_dir = "Client"; } else if (access("../Client/pack/Index", R_OK) == 0) { live_client_dir = "../Client"; } else if (access("../../Client/pack/Index", R_OK) == 0) { live_client_dir = "../../Client"; } #ifdef __ANDROID__ if (live_client_dir.empty()) { char* pref = SDL_GetPrefPath("mtgodot", "native-render"); if (pref) { const std::string candidate = std::string(pref) + "Client"; if (access((candidate + "/pack/Index").c_str(), R_OK) == 0) live_client_dir = candidate; SDL_free(pref); } } #endif if (!live_client_dir.empty() && !frames_specified) { interactive = true; } } if (interactive && !frames_specified) frames = 0; if (frames < 0 || (!interactive && frames == 0) || draw_count == 0 || draw_count > 4096 || triangles_per_draw == 0 || triangles_per_draw > 10000) throw std::runtime_error("invalid frame/draw/triangle count"); VulkanWindow renderer(vsync, init_width, init_height); if (mobile_mode) { renderer.touch_controller.set_enabled(true); renderer.touch_controller.update_screen_size(init_width, init_height); } if (!live_client_dir.empty()) { #ifdef MT_NATIVE_HAS_LIVE_CLIENT return run_live_client( renderer, live_client_dir, frames, fake_mobs, gpu_skinning, native_terrain, login_screen, live_server_spec, capture_out); #else throw std::runtime_error("mt_native_render was built without port_platform (--live-client unavailable)"); #endif } native_draw_capture::Capture capture{}; if (capture_path.empty()) { capture.draws = make_test_draws(draw_count, triangles_per_draw); } else { capture = native_draw_capture::read_capture(capture_path); } auto& draws = capture.draws; if (animate_first_draw && (draws.empty() || draws.front().positions.empty())) throw std::runtime_error("first draw has no geometry to animate"); const auto start = std::chrono::steady_clock::now(); int completed = 0; std::vector frame_ms; if (frames > 0) frame_ms.reserve(static_cast(frames)); while ((frames == 0 || completed < frames) && renderer.poll()) { const auto frame_start = std::chrono::steady_clock::now(); if (animate_first_draw && completed > 0) { draws.front().positions[0] += 0.0001f; ++draws.front().geometry_revision; } if (animate_bones && completed > 0) { const float offset = std::sin(float(completed) * 0.1f) * 0.5f; for (auto& d : draws) { for (std::size_t b = 12; b < d.bone_matrices.size(); b += 16) d.bone_matrices[b] += offset; } } renderer.render(draws, capture.textures, capture.ui_width, capture.ui_height, capture.ui_commands); frame_ms.push_back(std::chrono::duration( std::chrono::steady_clock::now() - frame_start).count()); ++completed; } renderer.finish_gpu_timings(); const auto ms = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); print_summary(renderer, completed, ms, 0.0, std::move(frame_ms)); return (frames == 0 || completed == frames) ? 0 : 2; } catch (const std::exception& error) { std::cerr << "native renderer: " << error.what() << '\n'; return 1; } }