Files
mtgodot-poc/native_render/main.cpp
T
shenleiandClaude Opus 5.5 0e8a1574cb render parity unit I: 40250 PythonGraphic SetGamma, cooldown fan, up/down buttons
SetGamma, RenderCoolTimeBox, RenderUp/DownButton and GetAvailableMemory are
transliterated from 40250 in place of stubs. The cooldown fan is emitted as
untextured polygon UI commands (native_render + python_ui_surface.gd draw
Bar+quad as a polygon); D3D8 shim gains SetGammaRamp/D3DGAMMARAMP.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-28 14:17:27 +09:00

3535 lines
178 KiB
C++

#include "RenderCommands3D.h"
#include "UIRenderCommands.h"
#include "draw_capture.h"
#include "dxt.h"
#include "touch_controller.h"
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
#include "platform/MilesLib/AudioCommands.h"
#include "platform/PackBackend.h"
#include "platform/ScriptLib/PythonBoot.h"
#include "platform/UserInterface/ServerClock.h"
#include "../tests/port_login_flow_server.h"
#include <thread>
#endif
#include <unistd.h>
#include <SDL3/SDL.h>
#include <SDL3/SDL_main.h>
#include <SDL3/SDL_vulkan.h>
#include <vulkan/vulkan.h>
#include "stb_image.h"
#ifdef __APPLE__
#include <AudioToolbox/AudioToolbox.h>
#endif
#include <algorithm>
#include <array>
#include <chrono>
#include <cctype>
#include <climits>
#include <cmath>
#include <cstdint>
#include <cstdlib>
#include <cstring>
#include <fstream>
#include <iostream>
#include <optional>
#include <sstream>
#include <stdexcept>
#include <string>
#include <unordered_map>
#include <unordered_set>
#include <vector>
namespace {
void check(VkResult result, const char* operation) {
if (result != VK_SUCCESS) throw std::runtime_error(std::string(operation) + ": VkResult " + std::to_string(result));
}
struct Vertex {
float position[3];
float normal[3];
float uv[2];
float color[4];
std::uint8_t joints[4];
float weights[4];
float mask_uv[2];
float rhw = 1.0f;
float vertex_fog = 1.0f;
};
struct PushConstants {
std::array<float, 16> mvp{};
// x: bone palette base + 1 (0 = not skinned), y: D3DFVF_XYZRHW, z: UI mask batch, w: unused.
std::array<float, 4> params{};
};
static_assert(sizeof(PushConstants) <= 128, "PushConstants must fit within Vulkan's 128-byte minimum guarantee");
// std430 payload selected with a dynamic storage-buffer offset for every draw: the D3D8
// fixed-function state in effect for the draw (texture stages, texture coordinate processing,
// lighting, material, fog, alpha test). A zero flags.x marks a 2D UI batch.
struct alignas(16) FixedFunctionState {
std::array<std::array<std::uint32_t, 4>, 2> stage_color{}; // op, arg1, arg2, D3DTSS_TEXCOORDINDEX
std::array<std::array<std::uint32_t, 4>, 2> stage_alpha{}; // op, arg1, arg2, D3DTSS_TEXTURETRANSFORMFLAGS
std::array<float, 4> fog_color{};
std::array<float, 4> fog_params{}; // start, end, density, range enabled
std::array<float, 4> texture_factor{};
std::array<float, 16> world_view{};
std::array<std::uint32_t, 4> flags{}; // fixed function, texture0, texture1, fog mode
std::array<float, 16> normal_matrix{}; // inverse transpose of world * view
std::array<std::uint32_t, 4> lighting_flags{}; // LIGHTING, COLORVERTEX, vertex has diffuse, NORMALIZENORMALS
std::array<std::uint32_t, 4> material_sources{}; // DIFFUSE, AMBIENT, EMISSIVE source, LOCALVIEWER
std::array<std::uint32_t, 4> alpha_test{}; // ALPHATESTENABLE, ALPHAFUNC, ALPHAREF, unused
std::array<float, 4> material_diffuse{};
std::array<float, 4> material_ambient{};
std::array<float, 4> material_emissive{};
std::array<float, 4> global_ambient{};
std::array<std::array<float, 16>, 2> texture_matrix{};
// Lights in camera space. position.w = D3DLIGHTTYPE (0 = disabled).
std::array<std::array<float, 4>, 8> light_position_type{};
std::array<std::array<float, 4>, 8> light_direction_range{};
std::array<std::array<float, 4>, 8> light_diffuse{};
std::array<std::array<float, 4>, 8> light_ambient{};
std::array<std::array<float, 4>, 8> light_attenuation{}; // a0, a1, a2, falloff
std::array<std::array<float, 4>, 8> light_spot{}; // cos(theta / 2), cos(phi / 2)
};
static_assert(sizeof(FixedFunctionState) == 1264);
constexpr std::array<float, 16> kIdentityMatrix = {
1.0f, 0.0f, 0.0f, 0.0f,
0.0f, 1.0f, 0.0f, 0.0f,
0.0f, 0.0f, 1.0f, 0.0f,
0.0f, 0.0f, 0.0f, 1.0f};
std::array<float, 16> multiply(const float* a, const float* b) {
std::array<float, 16> result{};
for (int row = 0; row < 4; ++row)
for (int col = 0; col < 4; ++col)
for (int k = 0; k < 4; ++k)
result[row * 4 + col] += a[row * 4 + k] * b[k * 4 + col];
return result;
}
std::array<float, 16> draw_mvp(const Render3DDraw& draw) {
const auto world_view = multiply(draw.world, draw.view);
return multiply(world_view.data(), draw.proj);
}
std::array<float, 4> unpack_argb(std::uint32_t argb) {
return {
float((argb >> 16) & 255) / 255.0f,
float((argb >> 8) & 255) / 255.0f,
float(argb & 255) / 255.0f,
float((argb >> 24) & 255) / 255.0f};
}
// Inverse transpose of the upper 3x3 of a row-vector matrix, the D3D8 normal transform.
std::array<float, 16> normal_matrix(const std::array<float, 16>& m) {
const float a = m[0], b = m[1], c = m[2], d = m[4], e = m[5], f = m[6], g = m[8], h = m[9], i = m[10];
const float A = e * i - f * h, B = -(d * i - f * g), C = d * h - e * g;
const float D = -(b * i - c * h), E = a * i - c * g, F = -(a * h - b * g);
const float G = b * f - c * e, H = -(a * f - c * d), I = a * e - b * d;
const float det = a * A + b * B + c * C;
std::array<float, 16> out = kIdentityMatrix;
if (det == 0.0f) return out;
const float s = 1.0f / det;
// inverse = adjugate / det; its transpose is the cofactor matrix / det.
out[0] = A * s; out[1] = B * s; out[2] = C * s;
out[4] = D * s; out[5] = E * s; out[6] = F * s;
out[8] = G * s; out[9] = H * s; out[10] = I * s;
return out;
}
PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded) {
PushConstants constants{};
constants.mvp = draw_mvp(draw);
constants.params = {skin_offset_encoded, draw.pretransformed ? 1.0f : 0.0f, 0.0f, 0.0f};
return constants;
}
FixedFunctionState make_fixed_function_state(const Render3DDraw& draw) {
FixedFunctionState state{};
for (int stage = 0; stage < 2; ++stage) {
state.stage_color[stage] = {draw.color_op[stage], draw.color_arg1[stage], draw.color_arg2[stage],
draw.texcoord_index[stage]};
state.stage_alpha[stage] = {draw.alpha_op[stage], draw.alpha_arg1[stage], draw.alpha_arg2[stage],
draw.texture_transform_flags[stage]};
std::copy(draw.texture_matrix[stage], draw.texture_matrix[stage] + 16, state.texture_matrix[stage].begin());
}
state.texture_factor = unpack_argb(draw.texture_factor);
state.world_view = multiply(draw.world, draw.view);
state.normal_matrix = normal_matrix(state.world_view);
state.fog_color = unpack_argb(draw.fog_color);
state.fog_params = {draw.fog_start, draw.fog_end, draw.fog_density,
draw.fog_range_enable ? 1.0f : 0.0f};
const std::uint32_t fog_mode = draw.fog_enable
? (draw.fog_table_mode ? draw.fog_table_mode : (draw.pretransformed ? 4u : draw.fog_vertex_mode)) : 0;
state.flags = {1u, !draw.texture0.empty() ? 1u : 0u, !draw.texture1.empty() ? 1u : 0u, fog_mode};
state.lighting_flags = {draw.lighting, draw.color_vertex, draw.diffuse.empty() ? 0u : 1u,
draw.normalize_normals};
state.material_sources = {draw.diffuse_material_source, draw.ambient_material_source,
draw.emissive_material_source, draw.local_viewer};
state.alpha_test = {draw.alpha_test, draw.alpha_func, draw.alpha_ref, 0};
std::copy(draw.material_diffuse, draw.material_diffuse + 4, state.material_diffuse.begin());
std::copy(draw.material_ambient, draw.material_ambient + 4, state.material_ambient.begin());
std::copy(draw.material_emissive, draw.material_emissive + 4, state.material_emissive.begin());
state.global_ambient = unpack_argb(draw.ambient);
const float* v = draw.view;
for (int i = 0; i < 8; ++i) {
const auto& light = draw.lights[i];
const float* p = light.position;
const float* d = light.direction;
std::array<float, 3> position{}, direction{};
for (int c = 0; c < 3; ++c) {
position[c] = p[0] * v[c] + p[1] * v[4 + c] + p[2] * v[8 + c] + v[12 + c];
direction[c] = d[0] * v[c] + d[1] * v[4 + c] + d[2] * v[8 + c];
}
const float length = std::sqrt(direction[0] * direction[0] + direction[1] * direction[1] +
direction[2] * direction[2]);
if (length > 0.0f)
for (float& c : direction) c /= length;
state.light_position_type[i] = {position[0], position[1], position[2], float(light.type)};
state.light_direction_range[i] = {direction[0], direction[1], direction[2], light.range};
std::copy(light.diffuse, light.diffuse + 4, state.light_diffuse[i].begin());
std::copy(light.ambient, light.ambient + 4, state.light_ambient[i].begin());
state.light_attenuation[i] = {light.attenuation[0], light.attenuation[1], light.attenuation[2], light.falloff};
state.light_spot[i] = {std::cos(light.theta * 0.5f), std::cos(light.phi * 0.5f), 0, 0};
}
return state;
}
std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size) {
const auto* data = static_cast<const std::uint8_t*>(bytes);
std::size_t i = 0;
for (; i + sizeof(std::uint64_t) <= size; i += sizeof(std::uint64_t)) {
std::uint64_t word = 0;
std::memcpy(&word, data + i, sizeof(word));
hash = (hash ^ word) * 1099511628211ull;
}
for (; i < size; ++i) hash = (hash ^ data[i]) * 1099511628211ull;
hash = (hash ^ size) * 1099511628211ull;
return hash;
}
// Immediate-mode draws have no source-buffer key. For them, verify content before reusing a slot.
std::uint64_t geometry_hash(const Render3DDraw& draw) {
std::uint64_t hash = 14695981039346656037ull;
const std::uint32_t flags = (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u);
hash = hash_bytes(hash, &flags, sizeof(flags));
hash = hash_bytes(hash, draw.positions.data(), draw.positions.size() * sizeof(float));
hash = hash_bytes(hash, draw.rhw.data(), draw.rhw.size() * sizeof(float));
hash = hash_bytes(hash, draw.vertex_fog.data(), draw.vertex_fog.size() * sizeof(float));
hash = hash_bytes(hash, draw.normals.data(), draw.normals.size() * sizeof(float));
hash = hash_bytes(hash, draw.uv0.data(), draw.uv0.size() * sizeof(float));
hash = hash_bytes(hash, draw.uv1.data(), draw.uv1.size() * sizeof(float));
hash = hash_bytes(hash, draw.diffuse.data(), draw.diffuse.size() * sizeof(std::uint32_t));
return hash_bytes(hash, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t));
}
mtgodot::Image decode_tga(const std::uint8_t* data, std::size_t size) {
mtgodot::Image out;
if (size < 18) return out;
const std::uint8_t id_len = data[0];
const std::uint8_t cmap_type = data[1];
const std::uint8_t img_type = data[2];
const std::uint32_t w = std::uint32_t(data[12]) | (std::uint32_t(data[13]) << 8);
const std::uint32_t h = std::uint32_t(data[14]) | (std::uint32_t(data[15]) << 8);
const std::uint8_t bpp = data[16];
const std::uint8_t desc = data[17];
if (cmap_type != 0 || !w || !h || w > 4096 || h > 4096) return out;
if (img_type != 2 && img_type != 3 && img_type != 10) return out;
const std::size_t bytes_per_pixel = bpp / 8;
if (bytes_per_pixel != 1 && bytes_per_pixel != 3 && bytes_per_pixel != 4) return out;
std::size_t offset = 18 + std::size_t(id_len);
if (offset > size) return out;
const std::size_t pixel_count = std::size_t(w) * std::size_t(h);
std::vector<std::uint8_t> temp(pixel_count * 4);
auto write_pixel = [&](std::size_t idx, const std::uint8_t* src) {
std::uint8_t* dst = &temp[idx * 4];
if (bytes_per_pixel == 1) {
dst[0] = dst[1] = dst[2] = src[0];
dst[3] = 255;
} else if (bytes_per_pixel == 3) {
dst[0] = src[2];
dst[1] = src[1];
dst[2] = src[0];
dst[3] = 255;
} else {
dst[0] = src[2];
dst[1] = src[1];
dst[2] = src[0];
dst[3] = src[3];
}
};
if (img_type == 2 || img_type == 3) {
if (offset + pixel_count * bytes_per_pixel > size) return out;
for (std::size_t i = 0; i < pixel_count; ++i)
write_pixel(i, data + offset + i * bytes_per_pixel);
} else if (img_type == 10) {
std::size_t i = 0;
while (i < pixel_count && offset < size) {
const std::uint8_t header = data[offset++];
const std::size_t run = (header & 0x7fu) + 1u;
if (header & 0x80u) {
if (offset + bytes_per_pixel > size) return out;
for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i)
write_pixel(i, data + offset);
offset += bytes_per_pixel;
} else {
if (offset + run * bytes_per_pixel > size) return out;
for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) {
write_pixel(i, data + offset);
offset += bytes_per_pixel;
}
}
}
if (i != pixel_count) return out;
}
out.w = static_cast<std::uint16_t>(w);
out.h = static_cast<std::uint16_t>(h);
const bool top_origin = (desc & 0x20u) != 0;
if (top_origin) {
out.rgba = std::move(temp);
} else {
out.rgba.resize(pixel_count * 4);
const std::size_t row_bytes = std::size_t(w) * 4;
for (std::uint32_t y = 0; y < h; ++y)
std::memcpy(&out.rgba[std::size_t(y) * row_bytes], &temp[std::size_t(h - 1 - y) * row_bytes], row_bytes);
}
return out;
}
mtgodot::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size) {
if (!data || size < 12) return {};
if (data[0] == 'M' && data[1] == 'T' && data[2] == 'R' && data[3] == 'A') {
std::uint32_t w = 0, h = 0;
std::memcpy(&w, data + 4, 4);
std::memcpy(&h, data + 8, 4);
const std::size_t bytes = std::size_t(w) * std::size_t(h) * 4;
if (w > 0 && h > 0 && w <= 4096 && h <= 4096 && size == 12 + bytes) {
mtgodot::Image out;
out.w = static_cast<std::uint16_t>(w);
out.h = static_cast<std::uint16_t>(h);
out.rgba.assign(data + 12, data + 12 + bytes);
return out;
}
return {};
}
if (data[0] == 'D' && data[1] == 'D' && data[2] == 'S' && data[3] == ' ')
return mtgodot::load_dds(data, size);
auto tga = decode_tga(data, size);
if (tga.ok()) return tga;
if (size > static_cast<std::size_t>(INT_MAX)) return {};
int w = 0, h = 0, channels = 0;
stbi_uc* pixels = stbi_load_from_memory(data, static_cast<int>(size), &w, &h, &channels, 4);
if (pixels && w > 0 && h > 0 && w <= 4096 && h <= 4096) {
mtgodot::Image out;
out.w = static_cast<std::uint16_t>(w);
out.h = static_cast<std::uint16_t>(h);
out.rgba.assign(pixels, pixels + std::size_t(w) * std::size_t(h) * 4);
stbi_image_free(pixels);
return out;
}
stbi_image_free(pixels);
return {};
}
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
bool read_live_pack_texture(const std::string& vpath, std::vector<std::uint8_t>& bytes) {
if (vpath.empty() || !mtpack40250::ready())
return false;
std::string norm = vpath;
for (char& ch : norm)
if (ch == '\\') ch = '/';
std::string stripped = norm;
if (stripped.size() >= 2 && stripped[1] == ':')
stripped = stripped.substr(2);
while (!stripped.empty() && stripped.front() == '/')
stripped.erase(stripped.begin());
auto lower = [](std::string s) {
for (char& ch : s)
if (ch >= 'A' && ch <= 'Z')
ch = static_cast<char>(ch - 'A' + 'a');
return s;
};
for (const std::string& candidate : {
norm, stripped, "d:/" + stripped,
lower(norm), lower(stripped), lower("d:/" + stripped)}) {
if (mtpack40250::read(candidate, bytes) && !bytes.empty())
return true;
}
return false;
}
SDL_Cursor* load_game_cursor(int shape) {
static constexpr std::array<const char*, 15> images = {
"cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub",
"cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub",
"cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub",
"cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"};
if (shape < 0 || shape >= static_cast<int>(images.size())) return nullptr;
const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape];
std::vector<std::uint8_t> sub_bytes;
if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr;
std::unordered_map<std::string, std::string> tokens;
std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end()));
std::string line;
while (std::getline(input, line)) {
std::istringstream fields(line);
std::string key, value;
if (!(fields >> key >> value)) continue;
if (!value.empty() && value.front() == '"') {
value.erase(0, 1);
if (!value.empty() && value.back() == '"') value.pop_back();
}
std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); });
tokens[key] = value;
}
if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr;
const std::string image_path = tokens["version"] == "2.0"
? std::string("d:/ymir work/ui/cursor/") + tokens["image"]
: std::string("d:/ymir work/ui/") + tokens["image"];
std::vector<std::uint8_t> image_bytes;
if (!read_live_pack_texture(image_path, image_bytes)) return nullptr;
const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size());
if (!image.ok()) return nullptr;
const int left = std::atoi(tokens["left"].c_str());
const int top = std::atoi(tokens["top"].c_str());
const int right = std::atoi(tokens["right"].c_str());
const int bottom = std::atoi(tokens["bottom"].c_str());
if (left < 0 || top < 0 || right <= left || bottom <= top ||
right > image.w || bottom > image.h) return nullptr;
const int width = right - left, height = bottom - top;
std::vector<std::uint8_t> pixels(std::size_t(width) * height * 4);
for (int y = 0; y < height; ++y)
std::memcpy(pixels.data() + std::size_t(y) * width * 4,
image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4,
std::size_t(width) * 4);
SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32,
pixels.data(), width * 4);
if (!surface) return nullptr;
const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0;
const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0;
SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y);
SDL_DestroySurface(surface);
return cursor;
}
class NativeAudioEngine {
struct MemoryAudioBuffer {
const std::uint8_t* data = nullptr;
std::size_t size = 0;
};
struct Voice {
const std::vector<float>* samples = nullptr;
std::size_t frame_cursor = 0;
float volume = 1.0f;
float target_volume = 1.0f;
float fade_step_per_frame = 0.0f;
bool stop_after_fade = false;
bool loop = false;
bool is_3d = false;
int handle = 0;
std::string filename;
};
#ifdef __APPLE__
static OSStatus mem_audio_read_proc(
void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) {
const auto* mem = static_cast<const MemoryAudioBuffer*>(inClientData);
if (inPosition < 0 || static_cast<std::size_t>(inPosition) >= mem->size) {
*actualCount = 0;
return noErr;
}
const std::size_t avail = mem->size - static_cast<std::size_t>(inPosition);
const std::size_t to_read = std::min<std::size_t>(requestCount, avail);
std::memcpy(buffer, mem->data + inPosition, to_read);
*actualCount = static_cast<UInt32>(to_read);
return noErr;
}
static SInt64 mem_audio_get_size_proc(void* inClientData) {
return static_cast<SInt64>(static_cast<const MemoryAudioBuffer*>(inClientData)->size);
}
#endif
public:
NativeAudioEngine() {
if (SDL_InitSubSystem(SDL_INIT_AUDIO)) {
SDL_AudioSpec spec{};
spec.format = SDL_AUDIO_F32;
spec.channels = 2;
spec.freq = 44100;
stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_PLAYBACK, &spec, nullptr, nullptr);
if (stream_)
SDL_ResumeAudioStreamDevice(stream_);
}
// Always DrainAudioCommands() from cold boot so stale queued commands don't accumulate.
(void)DrainAudioCommands();
}
~NativeAudioEngine() {
if (stream_)
SDL_DestroyAudioStream(stream_);
SDL_QuitSubSystem(SDL_INIT_AUDIO);
}
void pump() {
const auto commands = DrainAudioCommands();
for (const auto& cmd : commands) {
++commands_drained_;
switch (cmd.type) {
case AudioCommand::PlaySound2D: {
const auto* clip = get_clip(cmd.filename);
if (clip && !clip->empty()) {
Voice v{};
v.samples = clip;
v.volume = sound_volume_;
v.target_volume = sound_volume_;
v.filename = cmd.filename;
voices_.push_back(std::move(v));
++played_2d_;
}
break;
}
case AudioCommand::PlaySound3D: {
const auto* clip = get_clip(cmd.filename);
if (clip && !clip->empty()) {
Voice v{};
v.samples = clip;
v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
v.target_volume = v.volume;
v.loop = cmd.play_count != 1;
v.is_3d = true;
v.handle = cmd.id;
v.filename = cmd.filename;
voices_.push_back(std::move(v));
++played_3d_;
}
break;
}
case AudioCommand::StopSound3D:
voices_.erase(
std::remove_if(voices_.begin(), voices_.end(), [&](const Voice& v) {
return v.is_3d && v.handle == cmd.id;
}),
voices_.end());
break;
case AudioCommand::StopAllSound3D:
voices_.erase(
std::remove_if(voices_.begin(), voices_.end(), [](const Voice& v) { return v.is_3d; }),
voices_.end());
break;
case AudioCommand::SetSoundVolume3D:
for (auto& v : voices_) {
if (v.is_3d && v.handle == cmd.id) {
v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
v.target_volume = v.volume;
}
}
break;
case AudioCommand::PlayMusic:
case AudioCommand::FadeInMusic: {
const auto* clip = get_clip(cmd.filename);
if (clip && !clip->empty()) {
bgm_.samples = clip;
bgm_.frame_cursor = 0;
bgm_.loop = true;
bgm_.filename = cmd.filename;
bgm_.stop_after_fade = false;
const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
bgm_.target_volume = target;
if (cmd.type == AudioCommand::FadeInMusic) {
bgm_.volume = 0.0f;
bgm_.fade_step_per_frame = target / (44100.0f * 1.5f);
} else {
bgm_.volume = target;
bgm_.fade_step_per_frame = 0.0f;
}
++played_music_;
}
break;
}
case AudioCommand::FadeOutMusic:
case AudioCommand::FadeOutAllMusic:
if (bgm_.samples) {
bgm_.target_volume = 0.0f;
bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f);
bgm_.stop_after_fade = true;
}
break;
case AudioCommand::FadeLimitOutMusic:
if (bgm_.samples) {
bgm_.target_volume = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
bgm_.fade_step_per_frame = (bgm_.target_volume - bgm_.volume) / (44100.0f * 1.0f);
bgm_.stop_after_fade = false;
}
break;
case AudioCommand::SetMusicVolume:
music_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f);
if (bgm_.samples && !bgm_.stop_after_fade) {
bgm_.volume = music_volume_;
bgm_.target_volume = music_volume_;
}
break;
case AudioCommand::SetSoundVolume:
sound_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f);
break;
default:
break;
}
}
if (!stream_) return;
const int queued_bytes = SDL_GetAudioStreamAvailable(stream_);
if (queued_bytes < 0) return;
const std::size_t queued_frames = static_cast<std::size_t>(queued_bytes) / (2 * sizeof(float));
constexpr std::size_t kTargetQueuedFrames = 4410; // ~100 ms stereo @ 44.1 kHz
if (queued_frames >= kTargetQueuedFrames) return;
const std::size_t frames_to_mix = kTargetQueuedFrames - queued_frames;
mix_buffer_.assign(frames_to_mix * 2, 0.0f);
auto mix_voice = [&](Voice& v) -> bool {
if (!v.samples || v.samples->empty()) return false;
const std::size_t total_frames = v.samples->size() / 2;
if (total_frames == 0) return false;
const float* src = v.samples->data();
for (std::size_t f = 0; f < frames_to_mix; ++f) {
if (v.frame_cursor >= total_frames) {
if (v.loop) v.frame_cursor = 0;
else return false;
}
if (v.fade_step_per_frame != 0.0f) {
v.volume += v.fade_step_per_frame;
if ((v.fade_step_per_frame > 0.0f && v.volume >= v.target_volume) ||
(v.fade_step_per_frame < 0.0f && v.volume <= v.target_volume)) {
v.volume = v.target_volume;
v.fade_step_per_frame = 0.0f;
if (v.stop_after_fade && v.volume <= 0.0001f) {
v.samples = nullptr;
return false;
}
}
}
mix_buffer_[f * 2 + 0] += src[v.frame_cursor * 2 + 0] * v.volume;
mix_buffer_[f * 2 + 1] += src[v.frame_cursor * 2 + 1] * v.volume;
++v.frame_cursor;
}
return v.loop || v.frame_cursor < total_frames;
};
if (bgm_.samples) {
if (!mix_voice(bgm_)) bgm_.samples = nullptr;
}
for (auto it = voices_.begin(); it != voices_.end();) {
if (!mix_voice(*it)) it = voices_.erase(it);
else ++it;
}
for (float& sample : mix_buffer_)
sample = std::clamp(sample, -1.0f, 1.0f);
SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast<int>(mix_buffer_.size() * sizeof(float)));
}
private:
const std::vector<float>* get_clip(const std::string& vpath) {
if (vpath.empty()) return nullptr;
auto [it, inserted] = clips_.try_emplace(vpath);
if (!inserted) return &it->second;
#ifdef __APPLE__
std::vector<std::uint8_t> bytes;
if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16)
return &it->second;
MemoryAudioBuffer mem{bytes.data(), bytes.size()};
AudioFileID audio_file = nullptr;
if (AudioFileOpenWithCallbacks(
&mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr ||
!audio_file) {
return &it->second;
}
ExtAudioFileRef ext_file = nullptr;
if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) {
AudioFileClose(audio_file);
return &it->second;
}
AudioStreamBasicDescription client_format{};
client_format.mSampleRate = 44100.0;
client_format.mFormatID = kAudioFormatLinearPCM;
client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked;
client_format.mBytesPerPacket = 8;
client_format.mFramesPerPacket = 1;
client_format.mBytesPerFrame = 8;
client_format.mChannelsPerFrame = 2;
client_format.mBitsPerChannel = 32;
if (ExtAudioFileSetProperty(
ext_file,
kExtAudioFileProperty_ClientDataFormat,
sizeof(client_format),
&client_format) == noErr) {
std::vector<float> chunk(4096 * 2);
while (true) {
UInt32 frame_count = 4096;
AudioBufferList buf_list{};
buf_list.mNumberBuffers = 1;
buf_list.mBuffers[0].mNumberChannels = 2;
buf_list.mBuffers[0].mDataByteSize = static_cast<UInt32>(chunk.size() * sizeof(float));
buf_list.mBuffers[0].mData = chunk.data();
if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0)
break;
it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2);
if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes
break;
}
}
ExtAudioFileDispose(ext_file);
AudioFileClose(audio_file);
#endif
return &it->second;
}
SDL_AudioStream* stream_ = nullptr;
std::unordered_map<std::string, std::vector<float>> clips_;
std::vector<Voice> voices_;
Voice bgm_{};
std::vector<float> mix_buffer_;
float sound_volume_ = 0.8f;
float music_volume_ = 0.5f;
std::size_t commands_drained_ = 0;
std::size_t played_2d_ = 0;
std::size_t played_3d_ = 0;
std::size_t played_music_ = 0;
};
#endif
std::string bundled_file_path(const char* name) {
const char* base = SDL_GetBasePath();
return std::string(base ? base : "./") + name;
}
std::vector<std::uint32_t> read_spirv(const char* name) {
std::size_t size = 0;
void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size);
#ifdef __ANDROID__
if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size);
if (!bytes) bytes = SDL_LoadFile(name, &size);
#endif
if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name);
if (size == 0 || size % 4) {
SDL_free(bytes);
throw std::runtime_error(std::string("invalid SPIR-V size: ") + name);
}
std::vector<std::uint32_t> words(size / 4);
std::memcpy(words.data(), bytes, size);
SDL_free(bytes);
return words;
}
class VulkanWindow {
struct GeometryId {
std::uint64_t key = 0, signature = 0;
bool operator==(const GeometryId&) const = default;
};
struct GeometryIdHash {
std::size_t operator()(GeometryId id) const {
return std::size_t(id.key ^ (id.signature + 0x9e3779b97f4a7c15ull + (id.key << 6) + (id.key >> 2)));
}
};
struct Geometry {
VkBuffer buffer = VK_NULL_HANDLE;
VkDeviceMemory memory = VK_NULL_HANDLE;
VkDeviceSize index_offset = 0;
std::uint64_t last_used_frame = 0;
std::uint32_t vertex_count = 0, index_count = 0;
};
struct GpuTexture {
VkImage image = VK_NULL_HANDLE;
VkDeviceMemory memory = VK_NULL_HANDLE;
VkImageView view = VK_NULL_HANDLE;
VkDescriptorSet descriptor = VK_NULL_HANDLE;
VkDescriptorSet ui_descriptor = VK_NULL_HANDLE;
std::uint64_t last_used_frame = 0;
std::uint64_t last_decode_attempt_frame = 0;
};
struct PairedDescriptor {
VkDescriptorSet set = VK_NULL_HANDLE;
std::uint64_t last_used_frame = 0;
};
struct PreparedDraw {
Geometry* geometry = nullptr;
VkPipeline pipeline = VK_NULL_HANDLE;
VkDescriptorSet descriptor = VK_NULL_HANDLE;
PushConstants constants{};
FixedFunctionState fixed_state{};
std::uint32_t state_offset = 0;
VkViewport viewport{};
// IDirect3DDevice8::Clear recorded in draw order; geometry is null for these.
std::uint32_t clear_flags = 0;
VkClearValue clear_color{};
VkClearValue clear_depth{};
VkRect2D clear_rect{};
};
struct UiBatch {
std::uint32_t first_index = 0;
std::uint32_t index_count = 0;
VkPipeline pipeline = VK_NULL_HANDLE;
VkDescriptorSet descriptor = VK_NULL_HANDLE;
PushConstants constants{};
bool behind_3d = false;
std::uint32_t state_offset = 0;
};
static constexpr std::size_t kMaxBonesPerFrame = 65536;
static constexpr VkDeviceSize kBoneBufferBytes = kMaxBonesPerFrame * 16 * sizeof(float);
static constexpr std::size_t kMaxUiVerticesPerFrame = 65536;
static constexpr std::size_t kMaxUiIndicesPerFrame = 98304;
static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex);
static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t);
public:
struct Timings {
double sync_ms = 0;
double prepare_ms = 0;
double steady_prepare_ms = 0;
double submit_ms = 0;
double present_ms = 0;
double gpu_ms = 0;
std::uint64_t gpu_samples = 0;
};
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
void enable_game_hardware_cursor() {
hardware_cursor_enabled_ = true;
for (int shape = 0; shape < static_cast<int>(game_cursors_.size()); ++shape) {
game_cursors_[shape] = load_game_cursor(shape);
}
fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT);
SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_);
SDL_ShowCursor();
os_cursor_hidden_ = false;
}
void sync_game_cursor() {
if (!hardware_cursor_enabled_) return;
const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() &&
!PythonBoot::IsSoftwareCursorVisible();
if (visible == os_cursor_hidden_) {
if (visible) SDL_ShowCursor();
else SDL_HideCursor();
os_cursor_hidden_ = !visible;
}
if (!visible) return;
const int shape = PythonBoot::CursorShape();
if (shape == current_cursor_shape_) return;
SDL_Cursor* cursor = shape >= 0 && shape < static_cast<int>(game_cursors_.size())
? game_cursors_[shape] : nullptr;
if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_);
current_cursor_shape_ = shape;
}
#endif
explicit VulkanWindow(bool vsync = true, int init_width = 960, int init_height = 640) : vsync_(vsync) {
#ifdef __APPLE__
if (!std::getenv("VK_ICD_FILENAMES")) {
for (const char* icd : {
"/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json",
"/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json"}) {
if (std::ifstream(icd).good()) {
setenv("VK_ICD_FILENAMES", icd, 0);
break;
}
}
}
#endif
if (!SDL_Init(SDL_INIT_VIDEO)) throw std::runtime_error(SDL_GetError());
window_ = SDL_CreateWindow(
"Metin2 Native Vulkan Client",
std::max(init_width, 320),
std::max(init_height, 240),
SDL_WINDOW_VULKAN | SDL_WINDOW_RESIZABLE | SDL_WINDOW_HIGH_PIXEL_DENSITY);
if (!window_) throw std::runtime_error(SDL_GetError());
SDL_StartTextInput(window_);
Uint32 extension_count = 0;
const char* const* sdl_extensions = SDL_Vulkan_GetInstanceExtensions(&extension_count);
if (!sdl_extensions) throw std::runtime_error(SDL_GetError());
std::vector<const char*> extensions(sdl_extensions, sdl_extensions + extension_count);
std::uint32_t available_count = 0;
check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, nullptr), "vkEnumerateInstanceExtensionProperties count");
std::vector<VkExtensionProperties> available(available_count);
check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, available.data()), "vkEnumerateInstanceExtensionProperties");
const bool portability = std::any_of(available.begin(), available.end(), [](const auto& extension) {
return std::strcmp(extension.extensionName, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0;
});
if (portability && std::find_if(extensions.begin(), extensions.end(), [](const char* name) {
return std::strcmp(name, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0;
}) == extensions.end())
extensions.push_back(VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME);
VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO};
app.pApplicationName = "Metin2 native renderer";
app.apiVersion = VK_API_VERSION_1_1;
VkInstanceCreateInfo instance_info{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO};
instance_info.flags = portability ? VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR : 0;
instance_info.pApplicationInfo = &app;
instance_info.enabledExtensionCount = static_cast<std::uint32_t>(extensions.size());
instance_info.ppEnabledExtensionNames = extensions.data();
check(vkCreateInstance(&instance_info, nullptr, &instance_), "vkCreateInstance");
if (!SDL_Vulkan_CreateSurface(window_, instance_, nullptr, &surface_)) throw std::runtime_error(SDL_GetError());
select_device();
VkCommandPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO};
pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT;
pool_info.queueFamilyIndex = queue_family_;
check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool");
VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1;
check(vkAllocateCommandBuffers(device_, &alloc, &command_), "vkAllocateCommandBuffers");
VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO};
query_info.queryType = VK_QUERY_TYPE_TIMESTAMP;
query_info.queryCount = 2;
check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool");
create_swapchain();
create_descriptors_and_buffers();
create_render_pass_and_layout();
VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &acquire_), "vkCreateSemaphore acquire");
check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &rendered_), "vkCreateSemaphore rendered");
VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO};
fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT;
check(vkCreateFence(device_, &fence_info, nullptr, &fence_), "vkCreateFence");
touch_controller.update_screen_size(int(width()), int(height()));
}
~VulkanWindow() {
if (hardware_cursor_enabled_) {
SDL_SetCursor(nullptr);
for (SDL_Cursor* cursor : game_cursors_) if (cursor) SDL_DestroyCursor(cursor);
if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_);
}
if (device_) vkDeviceWaitIdle(device_);
if (bone_mapped_) vkUnmapMemory(device_, bone_memory_);
if (bone_buffer_) vkDestroyBuffer(device_, bone_buffer_, nullptr);
if (bone_memory_) vkFreeMemory(device_, bone_memory_, nullptr);
if (state_mapped_) vkUnmapMemory(device_, state_memory_);
if (state_buffer_) vkDestroyBuffer(device_, state_buffer_, nullptr);
if (state_memory_) vkFreeMemory(device_, state_memory_, nullptr);
if (ui_mapped_) vkUnmapMemory(device_, ui_memory_);
if (ui_buffer_) vkDestroyBuffer(device_, ui_buffer_, nullptr);
if (ui_memory_) vkFreeMemory(device_, ui_memory_, nullptr);
if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr);
if (fence_) vkDestroyFence(device_, fence_, nullptr);
if (acquire_) vkDestroySemaphore(device_, acquire_, nullptr);
if (rendered_) vkDestroySemaphore(device_, rendered_, nullptr);
for (auto& entry : geometries_) release_geometry(entry.second);
for (auto& entry : textures_) release_texture(entry.second);
release_texture(fallback_texture_);
for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr);
if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr);
if (fragment_module_) vkDestroyShaderModule(device_, fragment_module_, nullptr);
if (layout_) vkDestroyPipelineLayout(device_, layout_, nullptr);
if (descriptor_pool_) vkDestroyDescriptorPool(device_, descriptor_pool_, nullptr);
if (descriptor_layout_) vkDestroyDescriptorSetLayout(device_, descriptor_layout_, nullptr);
if (bone_descriptor_layout_) vkDestroyDescriptorSetLayout(device_, bone_descriptor_layout_, nullptr);
if (sampler_) vkDestroySampler(device_, sampler_, nullptr);
if (clamp_sampler_) vkDestroySampler(device_, clamp_sampler_, nullptr);
if (ui_sampler_) vkDestroySampler(device_, ui_sampler_, nullptr);
for (const auto& entry : state_samplers_) vkDestroySampler(device_, entry.second, nullptr);
for (auto framebuffer : framebuffers_) vkDestroyFramebuffer(device_, framebuffer, nullptr);
if (pass_) vkDestroyRenderPass(device_, pass_, nullptr);
if (depth_view_) vkDestroyImageView(device_, depth_view_, nullptr);
if (depth_image_) vkDestroyImage(device_, depth_image_, nullptr);
if (depth_memory_) vkFreeMemory(device_, depth_memory_, nullptr);
for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr);
if (swapchain_) vkDestroySwapchainKHR(device_, swapchain_, nullptr);
if (pool_) vkDestroyCommandPool(device_, pool_, nullptr);
if (device_) vkDestroyDevice(device_, nullptr);
if (surface_) vkDestroySurfaceKHR(instance_, surface_, nullptr);
if (instance_) vkDestroyInstance(instance_, nullptr);
if (window_) {
SDL_StopTextInput(window_);
SDL_DestroyWindow(window_);
}
SDL_Quit();
}
// 32-bit bottom-up BMP from a B8G8R8A8 / R8G8B8A8 swapchain readback.
void write_bmp(const std::string& path, const std::uint8_t* pixels) const {
const std::uint32_t w = extent_.width, h = extent_.height;
const bool rgba = swapchain_format_ == VK_FORMAT_R8G8B8A8_UNORM || swapchain_format_ == VK_FORMAT_R8G8B8A8_SRGB;
std::vector<std::uint8_t> file(54 + std::size_t(w) * h * 4);
auto put32 = [&](std::size_t at, std::uint32_t value) { std::memcpy(&file[at], &value, 4); };
file[0] = 'B'; file[1] = 'M';
put32(2, std::uint32_t(file.size())); put32(10, 54); put32(14, 40);
put32(18, w); put32(22, h); put32(26, 1u | (32u << 16)); put32(34, w * h * 4);
for (std::uint32_t y = 0; y < h; ++y)
for (std::uint32_t x = 0; x < w; ++x) {
const std::uint8_t* src = pixels + (std::size_t(y) * w + x) * 4;
std::uint8_t* dst = &file[54 + (std::size_t(h - 1 - y) * w + x) * 4];
dst[0] = rgba ? src[2] : src[0]; dst[1] = src[1]; dst[2] = rgba ? src[0] : src[2]; dst[3] = 255;
}
std::ofstream out(path, std::ios::binary);
out.write(reinterpret_cast<const char*>(file.data()), std::streamsize(file.size()));
}
void request_screenshot(std::string path, std::uint64_t frame) {
screenshot_path_ = std::move(path);
screenshot_frame_ = frame;
}
std::uint32_t width() const { return extent_.width; }
std::uint32_t height() const { return extent_.height; }
std::uint32_t logical_width() const {
int w = 0, h = 0;
if (window_) SDL_GetWindowSize(window_, &w, &h);
return w > 0 ? static_cast<std::uint32_t>(w) : extent_.width;
}
std::uint32_t logical_height() const {
int w = 0, h = 0;
if (window_) SDL_GetWindowSize(window_, &w, &h);
return h > 0 ? static_cast<std::uint32_t>(h) : extent_.height;
}
VkViewport full_viewport() const {
return {0, 0, float(extent_.width), float(extent_.height), 0, 1};
}
// D3DVIEWPORT8 (in the game's logical screen pixels) scaled to the swapchain extent.
VkViewport draw_viewport(const Render3DDraw& draw) const {
if (draw.viewport[2] <= 0 || draw.viewport[3] <= 0) return full_viewport();
const float sx = float(extent_.width) / float(logical_width());
const float sy = float(extent_.height) / float(logical_height());
return {draw.viewport[0] * sx, draw.viewport[1] * sy, draw.viewport[2] * sx, draw.viewport[3] * sy,
draw.viewport_z[0], draw.viewport_z[1]};
}
TouchController touch_controller;
static int sdl_scancode_to_dik(SDL_Scancode sc) {
switch (sc) {
case SDL_SCANCODE_ESCAPE: return 0x01;
case SDL_SCANCODE_1: return 0x02;
case SDL_SCANCODE_2: return 0x03;
case SDL_SCANCODE_3: return 0x04;
case SDL_SCANCODE_4: return 0x05;
case SDL_SCANCODE_5: return 0x06;
case SDL_SCANCODE_6: return 0x07;
case SDL_SCANCODE_7: return 0x08;
case SDL_SCANCODE_8: return 0x09;
case SDL_SCANCODE_9: return 0x0A;
case SDL_SCANCODE_0: return 0x0B;
case SDL_SCANCODE_MINUS: return 0x0C;
case SDL_SCANCODE_EQUALS: return 0x0D;
case SDL_SCANCODE_BACKSPACE: return 0x0E;
case SDL_SCANCODE_TAB: return 0x0F;
case SDL_SCANCODE_Q: return 0x10;
case SDL_SCANCODE_W: return 0x11;
case SDL_SCANCODE_E: return 0x12;
case SDL_SCANCODE_R: return 0x13;
case SDL_SCANCODE_T: return 0x14;
case SDL_SCANCODE_Y: return 0x15;
case SDL_SCANCODE_U: return 0x16;
case SDL_SCANCODE_I: return 0x17;
case SDL_SCANCODE_O: return 0x18;
case SDL_SCANCODE_P: return 0x19;
case SDL_SCANCODE_LEFTBRACKET: return 0x1A;
case SDL_SCANCODE_RIGHTBRACKET: return 0x1B;
case SDL_SCANCODE_RETURN: return 0x1C;
case SDL_SCANCODE_LCTRL: return 0x1D;
case SDL_SCANCODE_A: return 0x1E;
case SDL_SCANCODE_S: return 0x1F;
case SDL_SCANCODE_D: return 0x20;
case SDL_SCANCODE_F: return 0x21;
case SDL_SCANCODE_G: return 0x22;
case SDL_SCANCODE_H: return 0x23;
case SDL_SCANCODE_J: return 0x24;
case SDL_SCANCODE_K: return 0x25;
case SDL_SCANCODE_L: return 0x26;
case SDL_SCANCODE_SEMICOLON: return 0x27;
case SDL_SCANCODE_APOSTROPHE: return 0x28;
case SDL_SCANCODE_GRAVE: return 0x29;
case SDL_SCANCODE_LSHIFT: return 0x2A;
case SDL_SCANCODE_BACKSLASH: return 0x2B;
case SDL_SCANCODE_Z: return 0x2C;
case SDL_SCANCODE_X: return 0x2D;
case SDL_SCANCODE_C: return 0x2E;
case SDL_SCANCODE_V: return 0x2F;
case SDL_SCANCODE_B: return 0x30;
case SDL_SCANCODE_N: return 0x31;
case SDL_SCANCODE_M: return 0x32;
case SDL_SCANCODE_COMMA: return 0x33;
case SDL_SCANCODE_PERIOD: return 0x34;
case SDL_SCANCODE_SLASH: return 0x35;
case SDL_SCANCODE_RSHIFT: return 0x36;
case SDL_SCANCODE_KP_MULTIPLY: return 0x37;
case SDL_SCANCODE_LALT: return 0x38;
case SDL_SCANCODE_SPACE: return 0x39;
case SDL_SCANCODE_CAPSLOCK: return 0x3A;
case SDL_SCANCODE_F1: return 0x3B;
case SDL_SCANCODE_F2: return 0x3C;
case SDL_SCANCODE_F3: return 0x3D;
case SDL_SCANCODE_F4: return 0x3E;
case SDL_SCANCODE_F5: return 0x3F;
case SDL_SCANCODE_F6: return 0x40;
case SDL_SCANCODE_F7: return 0x41;
case SDL_SCANCODE_F8: return 0x42;
case SDL_SCANCODE_F9: return 0x43;
case SDL_SCANCODE_F10: return 0x44;
case SDL_SCANCODE_NUMLOCKCLEAR: return 0x45;
case SDL_SCANCODE_SCROLLLOCK: return 0x46;
case SDL_SCANCODE_KP_7: return 0x47;
case SDL_SCANCODE_KP_8: return 0x48;
case SDL_SCANCODE_KP_9: return 0x49;
case SDL_SCANCODE_KP_MINUS: return 0x4A;
case SDL_SCANCODE_KP_4: return 0x4B;
case SDL_SCANCODE_KP_5: return 0x4C;
case SDL_SCANCODE_KP_6: return 0x4D;
case SDL_SCANCODE_KP_PLUS: return 0x4E;
case SDL_SCANCODE_KP_1: return 0x4F;
case SDL_SCANCODE_KP_2: return 0x50;
case SDL_SCANCODE_KP_3: return 0x51;
case SDL_SCANCODE_KP_0: return 0x52;
case SDL_SCANCODE_KP_PERIOD: return 0x53;
case SDL_SCANCODE_F11: return 0x57;
case SDL_SCANCODE_F12: return 0x58;
case SDL_SCANCODE_KP_ENTER: return 0x9C;
case SDL_SCANCODE_RCTRL: return 0x9D;
case SDL_SCANCODE_KP_DIVIDE: return 0xB5;
case SDL_SCANCODE_RALT: return 0xB8;
case SDL_SCANCODE_HOME: return 0xC7;
case SDL_SCANCODE_UP: return 0xC8;
case SDL_SCANCODE_PAGEUP: return 0xC9;
case SDL_SCANCODE_LEFT: return 0xCB;
case SDL_SCANCODE_RIGHT: return 0xCD;
case SDL_SCANCODE_END: return 0xCF;
case SDL_SCANCODE_DOWN: return 0xD0;
case SDL_SCANCODE_PAGEDOWN: return 0xD1;
case SDL_SCANCODE_INSERT: return 0xD2;
case SDL_SCANCODE_DELETE: return 0xD3;
default: return 0;
}
}
static int sdl_scancode_to_vk(SDL_Scancode sc) {
switch (sc) {
case SDL_SCANCODE_BACKSPACE: return 0x08;
case SDL_SCANCODE_TAB: return 0x09;
case SDL_SCANCODE_RETURN:
case SDL_SCANCODE_KP_ENTER: return 0x0D;
case SDL_SCANCODE_ESCAPE: return 0x1B;
case SDL_SCANCODE_SPACE: return 0x20;
case SDL_SCANCODE_PAGEUP: return 0x21;
case SDL_SCANCODE_PAGEDOWN: return 0x22;
case SDL_SCANCODE_END: return 0x23;
case SDL_SCANCODE_HOME: return 0x24;
case SDL_SCANCODE_LEFT: return 0x25;
case SDL_SCANCODE_UP: return 0x26;
case SDL_SCANCODE_RIGHT: return 0x27;
case SDL_SCANCODE_DOWN: return 0x28;
case SDL_SCANCODE_INSERT: return 0x2D;
case SDL_SCANCODE_DELETE: return 0x2E;
case SDL_SCANCODE_F1: return 0x70;
case SDL_SCANCODE_F2: return 0x71;
case SDL_SCANCODE_F3: return 0x72;
case SDL_SCANCODE_F4: return 0x73;
case SDL_SCANCODE_F5: return 0x74;
case SDL_SCANCODE_F6: return 0x75;
case SDL_SCANCODE_F7: return 0x76;
case SDL_SCANCODE_F8: return 0x77;
case SDL_SCANCODE_F9: return 0x78;
case SDL_SCANCODE_F10: return 0x79;
case SDL_SCANCODE_F11: return 0x7A;
case SDL_SCANCODE_F12: return 0x7B;
default: return 0;
}
}
bool poll(bool forward_to_live_client = false) {
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
auto map_mouse = [&](float wx, float wy, int& ox, int& oy) {
int win_w = 0, win_h = 0;
SDL_GetWindowSize(window_, &win_w, &win_h);
unsigned ui_w = extent_.width ? extent_.width : 960;
unsigned ui_h = extent_.height ? extent_.height : 640;
UIRenderGetSize(&ui_w, &ui_h);
if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) {
ox = static_cast<int>(std::lround(wx * float(ui_w) / float(win_w)));
oy = static_cast<int>(std::lround(wy * float(ui_h) / float(win_h)));
} else {
ox = static_cast<int>(std::lround(wx));
oy = static_cast<int>(std::lround(wy));
}
};
std::optional<std::pair<int, int>> pending_mouse_move;
auto flush_mouse_move = [&] {
if (!pending_mouse_move) return;
const auto [mx, my] = *pending_mouse_move;
pending_mouse_move.reset();
last_mx_ = mx;
last_my_ = my;
if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return;
PythonBoot::UIMouseMove(mx, my);
};
#endif
SDL_Event event;
while (SDL_PollEvent(&event)) {
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) {
int mx = 0, my = 0;
map_mouse(event.motion.x, event.motion.y, mx, my);
pending_mouse_move = std::pair{mx, my};
continue;
}
// Preserve event order at button/key/focus boundaries while collapsing
// high-frequency motion into the most recent position.
flush_mouse_move();
#endif
if (event.type == SDL_EVENT_QUIT) return false;
if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) {
int ew = 0, eh = 0;
SDL_GetWindowSize(window_, &ew, &eh);
touch_controller.update_screen_size(ew, eh);
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
if (forward_to_live_client && ew > 0 && eh > 0) {
PythonBoot::SetUISize(ew, eh);
}
#endif
recreate_swapchain();
continue;
}
if (event.type == SDL_EVENT_FINGER_DOWN) {
touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y);
continue;
} else if (event.type == SDL_EVENT_FINGER_MOTION) {
touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y);
continue;
} else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) {
touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y);
continue;
}
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
if (forward_to_live_client) {
if (!hardware_cursor_enabled_ &&
(event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) {
std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n");
SDL_HideCursor();
os_cursor_hidden_ = true;
} else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) {
SDL_CaptureMouse(false);
PythonBoot::UIMouseButton(2, false, last_mx_, last_my_);
PythonBoot::UIMouseButton(3, false, last_mx_, last_my_);
PythonBoot::UIMouseButton(1, false, last_mx_, last_my_);
}
if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) {
const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN;
int btn = 0;
if (event.button.button == SDL_BUTTON_LEFT) btn = 1;
else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2;
else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3;
if (btn) {
int mx = 0, my = 0;
map_mouse(event.button.x, event.button.y, mx, my);
last_mx_ = mx;
last_my_ = my;
if (touch_controller.is_enabled()) {
if (touch_controller.on_mouse_button(btn, pressed, mx, my)) {
continue;
}
}
SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0);
PythonBoot::UIMouseButton(btn, pressed, mx, my);
}
} else if (event.type == SDL_EVENT_MOUSE_WHEEL) {
PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f));
} else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) {
const bool pressed = event.type == SDL_EVENT_KEY_DOWN;
if (pressed) {
if (const int vk = sdl_scancode_to_vk(event.key.scancode))
PythonBoot::UIIMEKeyDown(vk);
if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8);
else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9);
else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER)
PythonBoot::UIChar(13);
else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27);
}
if (!event.key.repeat) {
if (const int dik = sdl_scancode_to_dik(event.key.scancode))
PythonBoot::UIKey(dik, pressed);
}
} else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) {
const auto* s = reinterpret_cast<const unsigned char*>(event.text.text);
while (*s) {
unsigned cp = 0;
if (*s < 0x80u) {
cp = *s++;
} else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) {
cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu);
s += 2;
} else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) {
cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu);
s += 3;
} else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) {
cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu);
s += 4;
} else {
++s;
continue;
}
if (cp >= 32u && cp != 127u)
PythonBoot::UIChar(cp);
}
}
}
#else
(void)forward_to_live_client;
#endif
}
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
flush_mouse_move();
#endif
touch_controller.update();
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) {
SDL_HideCursor();
os_cursor_hidden_ = true;
}
#endif
return true;
}
void reset_timings() {
collect_pending_gpu_timestamp();
timings_ = {};
timed_frames_ = 0;
upload_count_ = 0;
uploaded_bytes_ = 0;
texture_upload_count_ = 0;
texture_uploaded_bytes_ = 0;
}
void finish_gpu_timings() {
if (!has_pending_query_) return;
check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences finish");
collect_pending_gpu_timestamp();
}
void render(
const std::vector<Render3DDraw>& draws,
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures,
std::uint32_t ui_width = 960,
std::uint32_t ui_height = 640,
const std::vector<UIRenderCommand>& ui_commands = {}) {
if (extent_.width == 0 || extent_.height == 0) return;
const auto start = std::chrono::steady_clock::now();
check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences");
collect_pending_gpu_timestamp();
prune_textures();
std::uint32_t image_index = 0;
const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, acquire_, VK_NULL_HANDLE, &image_index);
if (acquired == VK_ERROR_OUT_OF_DATE_KHR) {
recreate_swapchain();
return;
}
if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR");
const auto synchronized = std::chrono::steady_clock::now();
++frame_number_;
++timed_frames_;
std::vector<PreparedDraw> prepared_draws;
prepared_draws.reserve(draws.size());
std::unordered_set<std::uint64_t> active_keys;
std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0;
std::uint32_t bone_cursor = 0;
std::size_t required_bones = 0;
for (const auto& draw : draws) {
if (!draw.bone_matrices.empty()) {
if (draw.bone_matrices.size() % 16)
throw std::runtime_error("invalid bone palette length");
required_bones += draw.bone_matrices.size() / 16;
}
}
ensure_bone_capacity(required_bones);
for (const auto& draw : draws) {
if (draw.clear_flags) {
PreparedDraw clear{};
clear.clear_flags = draw.clear_flags;
const auto color = unpack_argb(draw.clear_color);
clear.clear_color.color = {{color[0], color[1], color[2], color[3]}};
clear.clear_depth.depthStencil = {draw.clear_z, 0};
// D3D8 Clear with no rectangles clears the current viewport.
const auto viewport = draw_viewport(draw);
const float x0 = std::clamp(viewport.x, 0.0f, float(extent_.width));
const float y0 = std::clamp(viewport.y, 0.0f, float(extent_.height));
const float x1 = std::clamp(viewport.x + viewport.width, 0.0f, float(extent_.width));
const float y1 = std::clamp(viewport.y + viewport.height, 0.0f, float(extent_.height));
clear.clear_rect = {{std::int32_t(x0), std::int32_t(y0)},
{std::uint32_t(x1 - x0), std::uint32_t(y1 - y0)}};
prepared_draws.push_back(clear);
continue;
}
if (draw.positions.empty() || draw.indices.empty()) continue;
const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw);
const GeometryId id{draw.geometry_key, signature};
auto [it, inserted] = geometries_.try_emplace(id);
auto& geometry = it->second;
if (inserted) {
upload_geometry(draw, geometry);
} else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) {
throw std::runtime_error("geometry key/revision reused with a different size");
}
geometry.last_used_frame = frame_number_;
if (draw.geometry_key) active_keys.insert(draw.geometry_key);
float skin_offset_encoded = 0.0f;
if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && bone_mapped_) {
const auto bone_count = static_cast<std::uint32_t>(draw.bone_matrices.size() / 16);
if (std::size_t(bone_cursor) + bone_count <= bone_capacity_) {
std::memcpy(
bone_mapped_ + std::size_t(bone_cursor) * 16,
draw.bone_matrices.data(),
draw.bone_matrices.size() * sizeof(float));
skin_offset_encoded = float(bone_cursor + 1u);
bone_cursor += bone_count;
++skinned_draw_count;
} else throw std::runtime_error("bone palette capacity exceeded");
}
const bool is_alpha = draw.alpha_blend != 0;
std::uint8_t cull = 0;
// D3D8 cull mode names describe the winding to discard. With the
// shader's Y flip, clockwise is Vulkan's front-facing winding.
if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise
else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise
const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0;
std::uint8_t blend = 0;
if (is_alpha) {
if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 &&
draw.dest_blend >= 1 && draw.dest_blend <= 11) {
blend = static_cast<std::uint8_t>((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4));
} else {
blend = 1;
}
}
const auto pipeline = get_pipeline(cull, depth, blend, draw.lines,
static_cast<std::uint8_t>(draw.z_func));
// D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples.
const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1;
const auto descriptor = get_texture_descriptor(
draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw);
auto constants = make_push_constants(draw, skin_offset_encoded);
if (draw.pretransformed) {
const float screen_width = float(logical_width());
const float screen_height = float(logical_height());
// Screen pixels (y down) to D3D NDC (y up); the shader's Y flip then maps to Vulkan.
constants.mvp = {2.0f / screen_width, 0, 0, 0,
0, -2.0f / screen_height, 0, 0,
0, 0, 1, 0,
-1, 1, 0, 1};
}
PreparedDraw prepared_draw{};
prepared_draw.geometry = &geometry;
prepared_draw.pipeline = pipeline;
prepared_draw.descriptor = descriptor;
prepared_draw.constants = constants;
prepared_draw.fixed_state = make_fixed_function_state(draw);
// XYZRHW vertices are already in screen space and bypass the viewport transform.
prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw);
prepared_draws.push_back(prepared_draw);
vertex_count += geometry.vertex_count;
index_count += geometry.index_count;
}
for (auto it = geometries_.begin(); it != geometries_.end();) {
const auto age = frame_number_ - it->second.last_used_frame;
const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key);
if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) {
release_geometry(it->second);
it = geometries_.erase(it);
} else ++it;
}
std::vector<UiBatch> ui_batches;
std::size_t ui_quad_count = 0;
if (ui_mapped_) {
const uint32_t target_ui_w = ui_width ? ui_width : 960;
const uint32_t target_ui_h = ui_height ? ui_height : 640;
if (touch_controller.is_enabled()) {
touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h));
std::vector<UIRenderCommand> combined_ui = ui_commands;
touch_controller.append_ui_commands(combined_ui);
if (!combined_ui.empty()) {
build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count);
}
} else if (!ui_commands.empty()) {
build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count);
}
}
ensure_state_capacity(prepared_draws.size() + ui_batches.size());
std::size_t state_index = 0;
for (auto& draw : prepared_draws) {
if (!draw.geometry) continue;
draw.state_offset = static_cast<std::uint32_t>(state_index * state_stride_);
std::memcpy(state_mapped_ + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState));
++state_index;
}
const FixedFunctionState ui_state{};
for (auto& batch : ui_batches) {
batch.state_offset = static_cast<std::uint32_t>(state_index * state_stride_);
std::memcpy(state_mapped_ + batch.state_offset, &ui_state, sizeof(FixedFunctionState));
++state_index;
}
const auto prepared = std::chrono::steady_clock::now();
check(vkResetFences(device_, 1, &fence_), "vkResetFences");
check(vkResetCommandBuffer(command_, 0), "vkResetCommandBuffer");
VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
check(vkBeginCommandBuffer(command_, &begin), "vkBeginCommandBuffer");
vkCmdResetQueryPool(command_, query_pool_, 0, 2);
vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, 0);
VkClearValue clears[2]{};
// 40250 clears the back buffer to 0xff000000 once at device creation and afterwards
// only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer).
clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}};
clears[1].depthStencil = {1.0f, 0};
VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO};
pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index);
pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = 2; pass_begin.pClearValues = clears;
vkCmdBeginRenderPass(command_, &pass_begin, VK_SUBPASS_CONTENTS_INLINE);
VkPipeline current_pipeline = VK_NULL_HANDLE;
VkDescriptorSet current_descriptor = VK_NULL_HANDLE;
VkViewport current_viewport{-1, -1, -1, -1, -1, -1};
auto set_viewport = [&](const VkViewport& viewport) {
if (std::memcmp(&viewport, &current_viewport, sizeof(viewport)) == 0) return;
vkCmdSetViewport(command_, 0, 1, &viewport);
current_viewport = viewport;
};
auto record_ui_pass = [&](bool behind_3d) {
bool ui_bound = false;
set_viewport(full_viewport());
for (const auto& batch : ui_batches) {
if (batch.behind_3d != behind_3d) continue;
if (!ui_bound) {
const VkDeviceSize v_offset = 0;
vkCmdBindVertexBuffers(command_, 0, 1, &ui_buffer_, &v_offset);
vkCmdBindIndexBuffer(command_, ui_buffer_, ui_vertex_bytes_, VK_INDEX_TYPE_UINT32);
ui_bound = true;
}
if (batch.pipeline != current_pipeline) {
vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline);
current_pipeline = batch.pipeline;
}
if (batch.descriptor != current_descriptor) {
vkCmdBindDescriptorSets(
command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr);
current_descriptor = batch.descriptor;
}
vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1,
&bone_descriptor_set_, 1, &batch.state_offset);
vkCmdPushConstants(
command_,
layout_,
VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT,
0,
sizeof(batch.constants),
&batch.constants);
vkCmdDrawIndexed(command_, batch.index_count, 1, batch.first_index, 0, 0);
}
};
record_ui_pass(true);
for (const auto& prepared_draw : prepared_draws) {
if (!prepared_draw.geometry) {
VkClearAttachment attachments[2]{};
std::uint32_t attachment_count = 0;
if (prepared_draw.clear_flags & 1u) { // D3DCLEAR_TARGET
attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
attachments[attachment_count].colorAttachment = 0;
attachments[attachment_count].clearValue = prepared_draw.clear_color;
++attachment_count;
}
if (prepared_draw.clear_flags & 2u) { // D3DCLEAR_ZBUFFER
attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
attachments[attachment_count].clearValue = prepared_draw.clear_depth;
++attachment_count;
}
const VkClearRect rect{prepared_draw.clear_rect, 0, 1};
if (attachment_count && rect.rect.extent.width && rect.rect.extent.height)
vkCmdClearAttachments(command_, attachment_count, attachments, 1, &rect);
continue;
}
set_viewport(prepared_draw.viewport);
if (prepared_draw.pipeline != current_pipeline) {
vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline);
current_pipeline = prepared_draw.pipeline;
}
if (prepared_draw.descriptor != current_descriptor) {
vkCmdBindDescriptorSets(
command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr);
current_descriptor = prepared_draw.descriptor;
}
vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1,
&bone_descriptor_set_, 1, &prepared_draw.state_offset);
const auto& geometry = *prepared_draw.geometry;
const VkDeviceSize vertex_offset = 0;
vkCmdBindVertexBuffers(command_, 0, 1, &geometry.buffer, &vertex_offset);
vkCmdBindIndexBuffer(command_, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32);
vkCmdPushConstants(
command_,
layout_,
VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT,
0,
sizeof(prepared_draw.constants),
&prepared_draw.constants);
vkCmdDrawIndexed(command_, geometry.index_count, 1, 0, 0, 0);
}
record_ui_pass(false);
vkCmdEndRenderPass(command_);
VkBuffer readback = VK_NULL_HANDLE;
VkDeviceMemory readback_memory = VK_NULL_HANDLE;
const bool screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_;
if (screenshot) {
const VkDeviceSize bytes = VkDeviceSize(extent_.width) * extent_.height * 4;
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
info.size = bytes; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
check(vkCreateBuffer(device_, &info, nullptr, &readback), "vkCreateBuffer readback");
VkMemoryRequirements requirements{};
vkGetBufferMemoryRequirements(device_, readback, &requirements);
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
allocation.allocationSize = requirements.size;
allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits,
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
check(vkAllocateMemory(device_, &allocation, nullptr, &readback_memory), "vkAllocateMemory readback");
check(vkBindBufferMemory(device_, readback, readback_memory, 0), "vkBindBufferMemory readback");
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
barrier.oldLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
barrier.image = swapchain_images_.at(image_index);
barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
vkCmdPipelineBarrier(command_, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT,
VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier);
VkBufferImageCopy region{};
region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1};
region.imageExtent = {extent_.width, extent_.height, 1};
vkCmdCopyImageToBuffer(command_, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, &region);
barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
barrier.dstAccessMask = 0;
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
vkCmdPipelineBarrier(command_, VK_PIPELINE_STAGE_TRANSFER_BIT,
VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier);
}
vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, 1);
check(vkEndCommandBuffer(command_), "vkEndCommandBuffer");
has_pending_query_ = true;
const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT;
VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO};
submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &acquire_; submit.pWaitDstStageMask = &stage;
submit.commandBufferCount = 1; submit.pCommandBuffers = &command_;
submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_;
check(vkQueueSubmit(queue_, 1, &submit, fence_), "vkQueueSubmit");
if (screenshot) {
check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences readback");
void* mapped = nullptr;
check(vkMapMemory(device_, readback_memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback");
write_bmp(screenshot_path_, static_cast<const std::uint8_t*>(mapped));
vkUnmapMemory(device_, readback_memory);
vkDestroyBuffer(device_, readback, nullptr);
vkFreeMemory(device_, readback_memory, nullptr);
}
const auto submitted = std::chrono::steady_clock::now();
VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR};
present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_;
present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index;
const auto presented = vkQueuePresentKHR(queue_, &present);
if (presented == VK_ERROR_OUT_OF_DATE_KHR || presented == VK_SUBOPTIMAL_KHR) {
recreate_swapchain();
} else if (presented != VK_SUCCESS) {
check(presented, "vkQueuePresentKHR");
}
const auto presented_at = std::chrono::steady_clock::now();
const auto prepare_ms = std::chrono::duration<double, std::milli>(prepared - synchronized).count();
timings_.sync_ms += std::chrono::duration<double, std::milli>(synchronized - start).count();
timings_.prepare_ms += prepare_ms;
if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms;
timings_.submit_ms += std::chrono::duration<double, std::milli>(submitted - prepared).count();
timings_.present_ms += std::chrono::duration<double, std::milli>(presented_at - submitted).count();
last_draw_count_ = prepared_draws.size();
last_skinned_draw_count_ = skinned_draw_count;
last_ui_batch_count_ = ui_batches.size();
last_ui_quad_count_ = ui_quad_count;
last_vertex_count_ = vertex_count;
last_index_count_ = index_count;
}
std::size_t draw_count() const { return last_draw_count_; }
std::size_t skinned_draw_count() const { return last_skinned_draw_count_; }
std::size_t ui_batch_count() const { return last_ui_batch_count_; }
std::size_t ui_quad_count() const { return last_ui_quad_count_; }
std::size_t vertex_count() const { return last_vertex_count_; }
std::size_t index_count() const { return last_index_count_; }
std::size_t upload_count() const { return upload_count_; }
std::size_t uploaded_bytes() const { return uploaded_bytes_; }
std::size_t texture_upload_count() const { return texture_upload_count_; }
std::size_t texture_uploaded_bytes() const { return texture_uploaded_bytes_; }
const std::string& device_name() const { return device_name_; }
const char* present_mode_name() const {
switch (present_mode_) {
case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE";
case VK_PRESENT_MODE_MAILBOX_KHR: return "MAILBOX";
default: return "FIFO";
}
}
Timings timings() const { return timings_; }
private:
void recreate_swapchain() {
int w = 0, h = 0;
SDL_GetWindowSizeInPixels(window_, &w, &h);
if (w <= 0 || h <= 0) return;
vkDeviceWaitIdle(device_);
for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr);
pipelines_.clear();
for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr);
framebuffers_.clear();
if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; }
if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; }
if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; }
for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr);
image_views_.clear();
if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; }
create_swapchain();
for (auto view : image_views_) {
VkImageView fb_attachments[2] = {view, depth_view_};
VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO};
framebuffer.renderPass = pass_; framebuffer.attachmentCount = 2; framebuffer.pAttachments = fb_attachments;
framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1;
VkFramebuffer handle = VK_NULL_HANDLE;
check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer recreate");
framebuffers_.push_back(handle);
}
get_pipeline(0, 2, 0);
}
void collect_pending_gpu_timestamp() {
if (!has_pending_query_) return;
has_pending_query_ = false;
std::uint64_t timestamps[2] = {};
const VkResult res = vkGetQueryPoolResults(
device_, query_pool_, 0, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), VK_QUERY_RESULT_64_BIT);
if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) {
const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_);
timings_.gpu_ms += ns * 1e-6;
++timings_.gpu_samples;
}
}
void build_ui_batches(
std::uint32_t ui_width,
std::uint32_t ui_height,
const std::vector<UIRenderCommand>& commands,
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures,
std::vector<UiBatch>& batches,
std::size_t& out_quad_count) {
ensure_ui_capacity(commands.size());
auto* vertices = reinterpret_cast<Vertex*>(ui_mapped_);
auto* indices = reinterpret_cast<std::uint32_t*>(ui_mapped_ + ui_vertex_bytes_);
std::uint32_t v_count = 0;
std::uint32_t i_count = 0;
const float inv_w = 2.0f / float(ui_width);
const float inv_h = 2.0f / float(ui_height);
for (const auto& cmd : commands) {
if (v_count + 4 > ui_vertex_capacity_ || i_count + 6 > ui_index_capacity_)
throw std::runtime_error("UI command buffer capacity exceeded");
if (cmd.kind == UIRenderCommand::Text) continue;
float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2;
float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev;
const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1;
float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {};
std::array<float, 4> top_color = unpack_argb(cmd.argb);
std::array<float, 4> bot_color =
cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color;
if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue;
if (cmd.kind == UIRenderCommand::Line) {
const float dx = x2 - x1, dy = y2 - y1;
const float len = std::sqrt(dx * dx + dy * dy);
if (len < 0.1f) continue;
const float nx = -dy / len * 0.5f;
const float ny = dx / len * 0.5f;
qx[0] = x1 + nx; qy[0] = y1 + ny;
qx[1] = x2 + nx; qy[1] = y2 + ny;
qx[2] = x1 - nx; qy[2] = y1 - ny;
qx[3] = x2 - nx; qy[3] = y2 - ny;
for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; }
} else if (cmd.kind == UIRenderCommand::Image && cmd.quad) {
const bool axis_aligned =
std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f &&
std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f &&
std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f &&
std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f;
for (int k = 0; k < 4; ++k) {
mu[k] = cmd.mu[k];
mv[k] = cmd.mv[k];
}
if (has_clip && axis_aligned) {
const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3];
const float cx1 = std::max(ox1, cmd.clip_x1);
const float cy1 = std::max(oy1, cmd.clip_y1);
const float cx2 = std::min(ox2, cmd.clip_x2);
const float cy2 = std::min(oy2, cmd.clip_y2);
if (cx2 <= cx1 || cy2 <= cy1) continue;
float msu = cmd.mu[0], meu = cmd.mu[1];
float msv = cmd.mv[0], mev = cmd.mv[2];
if (ox2 > ox1) {
const float t0 = (cx1 - ox1) / (ox2 - ox1);
const float t1 = (cx2 - ox1) / (ox2 - ox1);
const float du = eu - su;
su = cmd.su + du * t0;
eu = cmd.su + du * t1;
const float mdu = meu - msu;
msu = cmd.mu[0] + mdu * t0;
meu = cmd.mu[0] + mdu * t1;
}
if (oy2 > oy1) {
const float t0 = (cy1 - oy1) / (oy2 - oy1);
const float t1 = (cy2 - oy1) / (oy2 - oy1);
const float dv = ev - sv;
sv = cmd.sv + dv * t0;
ev = cmd.sv + dv * t1;
const float mdv = mev - msv;
msv = cmd.mv[0] + mdv * t0;
mev = cmd.mv[0] + mdv * t1;
}
qx[0] = cx1; qy[0] = cy1;
qx[1] = cx2; qy[1] = cy1;
qx[2] = cx1; qy[2] = cy2;
qx[3] = cx2; qy[3] = cy2;
mu[0] = msu; mv[0] = msv;
mu[1] = meu; mv[1] = msv;
mu[2] = msu; mv[2] = mev;
mu[3] = meu; mv[3] = mev;
} else {
if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2))
continue;
for (int k = 0; k < 4; ++k) {
qx[k] = cmd.qx[k];
qy[k] = cmd.qy[k];
}
}
u[0] = su; v[0] = sv;
u[1] = eu; v[1] = sv;
u[2] = su; v[2] = ev;
u[3] = eu; v[3] = ev;
} else if (cmd.kind == UIRenderCommand::Bar && cmd.quad) {
// An untextured polygon piece (CPythonGraphic::RenderCoolTimeBox's fan triangles).
if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2))
continue;
for (int k = 0; k < 4; ++k) {
qx[k] = cmd.qx[k];
qy[k] = cmd.qy[k];
u[k] = 0.0f;
v[k] = 0.0f;
}
} else {
if (has_clip) {
x1 = std::max(x1, cmd.clip_x1);
y1 = std::max(y1, cmd.clip_y1);
x2 = std::min(x2, cmd.clip_x2);
y2 = std::min(y2, cmd.clip_y2);
}
if (x2 <= x1 || y2 <= y1) continue;
qx[0] = x1; qy[0] = y1;
qx[1] = x2; qy[1] = y1;
qx[2] = x1; qy[2] = y2;
qx[3] = x2; qy[3] = y2;
u[0] = su; v[0] = sv;
u[1] = eu; v[1] = sv;
u[2] = su; v[2] = ev;
u[3] = eu; v[3] = ev;
}
const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string();
const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string();
const bool is_masked = !mask_name.empty();
// CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use
// INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR.
const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image
? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1))
: 1;
const VkPipeline pipeline = get_pipeline(0, 0, blend_mode);
const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures);
for (int k = 0; k < 4; ++k) {
Vertex& vert = vertices[v_count + k];
vert.position[0] = qx[k] * inv_w - 1.0f;
vert.position[1] = 1.0f - qy[k] * inv_h;
vert.position[2] = 0.0f;
vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f;
vert.uv[0] = u[k]; vert.uv[1] = v[k];
const auto& col = (k < 2) ? top_color : bot_color;
vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3];
vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0;
vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f;
vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k];
vert.rhw = 1.0f;
}
indices[i_count + 0] = v_count + 0;
indices[i_count + 1] = v_count + 1;
indices[i_count + 2] = v_count + 2;
indices[i_count + 3] = v_count + 2;
indices[i_count + 4] = v_count + 1;
indices[i_count + 3 + 2] = v_count + 3;
const float mask_flag = is_masked ? 1.0f : 0.0f;
if (!batches.empty() &&
batches.back().behind_3d == cmd.behind_3d &&
batches.back().pipeline == pipeline &&
batches.back().descriptor == descriptor &&
batches.back().constants.params[2] == mask_flag) {
batches.back().index_count += 6;
} else {
UiBatch batch{};
batch.first_index = i_count;
batch.index_count = 6;
batch.pipeline = pipeline;
batch.descriptor = descriptor;
batch.behind_3d = cmd.behind_3d;
batch.constants.mvp = kIdentityMatrix;
batch.constants.params = {0.0f, 0.0f, mask_flag, 0.0f};
batches.push_back(batch);
}
v_count += 4;
i_count += 6;
++out_quad_count;
}
}
std::uint32_t find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const {
VkPhysicalDeviceMemoryProperties properties{};
vkGetPhysicalDeviceMemoryProperties(physical_, &properties);
for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i)
if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags)
return i;
throw std::runtime_error("no matching Vulkan memory type");
}
void select_device() {
std::uint32_t count = 0;
check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count");
if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK");
std::vector<VkPhysicalDevice> candidates(count);
check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices");
for (auto candidate : candidates) {
std::uint32_t family_count = 0;
vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr);
std::vector<VkQueueFamilyProperties> families(family_count);
vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data());
for (std::uint32_t i = 0; i < family_count; ++i) {
VkBool32 present = VK_FALSE;
check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support");
if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) {
physical_ = candidate; queue_family_ = i; break;
}
}
if (physical_) break;
}
if (!physical_) throw std::runtime_error("no graphics/present queue family");
VkPhysicalDeviceProperties properties{};
vkGetPhysicalDeviceProperties(physical_, &properties);
device_name_ = properties.deviceName;
timestamp_period_ns_ = properties.limits.timestampPeriod;
VkPhysicalDeviceFeatures supported_features{};
vkGetPhysicalDeviceFeatures(physical_, &supported_features);
VkPhysicalDeviceFeatures enabled_features{};
if (supported_features.samplerAnisotropy) {
enabled_features.samplerAnisotropy = VK_TRUE;
max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy);
}
std::uint32_t extension_count = 0;
check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count");
std::vector<VkExtensionProperties> available(extension_count);
check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions");
std::vector<const char*> extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME};
constexpr const char* portability_subset = "VK_KHR_portability_subset";
for (const auto& extension : available)
if (std::strcmp(extension.extensionName, portability_subset) == 0)
extensions.push_back(portability_subset);
const float priority = 1.0f;
VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO};
queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority;
VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO};
device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info;
device_info.enabledExtensionCount = static_cast<std::uint32_t>(extensions.size());
device_info.ppEnabledExtensionNames = extensions.data();
device_info.pEnabledFeatures = &enabled_features;
check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice");
vkGetDeviceQueue(device_, queue_family_, 0, &queue_);
}
void create_swapchain() {
VkSurfaceCapabilitiesKHR capabilities{};
check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities");
std::uint32_t count = 0;
check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count");
if (!count) throw std::runtime_error("surface has no formats");
std::vector<VkSurfaceFormatKHR> formats(count);
check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats");
VkSurfaceFormatKHR format = formats.front();
for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate;
swapchain_format_ = format.format;
int width = 0, height = 0;
SDL_GetWindowSizeInPixels(window_, &width, &height);
if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent;
else {
extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width);
extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height);
}
present_mode_ = VK_PRESENT_MODE_FIFO_KHR;
if (!vsync_) {
std::uint32_t mode_count = 0;
check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count");
std::vector<VkPresentModeKHR> modes(mode_count);
if (mode_count)
check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes");
if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end())
present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR;
else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end())
present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR;
}
VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR};
info.surface = surface_;
info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX);
info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_;
info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT |
(capabilities.supportedUsageFlags & VK_IMAGE_USAGE_TRANSFER_SRC_BIT);
info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; info.preTransform = capabilities.currentTransform;
info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_;
info.clipped = VK_TRUE;
check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR");
check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count");
std::vector<VkImage> images(count);
check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images");
swapchain_images_ = images;
for (auto image : images) {
VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_;
view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1;
VkImageView handle = VK_NULL_HANDLE;
check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView");
image_views_.push_back(handle);
}
VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
depth_info.imageType = VK_IMAGE_TYPE_2D;
depth_info.format = VK_FORMAT_D32_SFLOAT;
depth_info.extent = {extent_.width, extent_.height, 1};
depth_info.mipLevels = 1; depth_info.arrayLayers = 1;
depth_info.samples = VK_SAMPLE_COUNT_1_BIT;
depth_info.tiling = VK_IMAGE_TILING_OPTIMAL;
depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth");
VkMemoryRequirements depth_reqs{};
vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs);
VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
depth_alloc.allocationSize = depth_reqs.size;
depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth");
check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth");
VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D;
depth_view.format = VK_FORMAT_D32_SFLOAT;
depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1;
check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth");
}
void create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) {
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
info.size = size;
info.usage = usage;
info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host");
VkMemoryRequirements reqs{};
vkGetBufferMemoryRequirements(device_, buffer, &reqs);
VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
alloc.allocationSize = reqs.size;
alloc.memoryTypeIndex = find_memory_type(
reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host");
check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host");
check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host");
}
void create_descriptors_and_buffers() {
VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO};
sampler_info.magFilter = VK_FILTER_LINEAR;
sampler_info.minFilter = VK_FILTER_LINEAR;
sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR;
sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT;
sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT;
sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT;
sampler_info.minLod = 0.0f;
sampler_info.maxLod = VK_LOD_CLAMP_NONE;
if (max_anisotropy_ > 1.0f) {
sampler_info.anisotropyEnable = VK_TRUE;
sampler_info.maxAnisotropy = max_anisotropy_;
}
check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler");
VkSamplerCreateInfo clamp_info = sampler_info;
clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp");
VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO};
ui_sampler_info.magFilter = VK_FILTER_LINEAR;
ui_sampler_info.minFilter = VK_FILTER_LINEAR;
ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST;
ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
ui_sampler_info.minLod = 0.0f;
ui_sampler_info.maxLod = 0.0f;
ui_sampler_info.anisotropyEnable = VK_FALSE;
ui_sampler_info.maxAnisotropy = 1.0f;
check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui");
VkDescriptorSetLayoutBinding tex_bindings[2]{};
tex_bindings[0].binding = 0;
tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
tex_bindings[0].descriptorCount = 1;
tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT;
tex_bindings[1].binding = 1;
tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
tex_bindings[1].descriptorCount = 1;
tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT;
VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO};
layout_info.bindingCount = 2;
layout_info.pBindings = tex_bindings;
check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex");
VkDescriptorSetLayoutBinding bone_bindings[2]{};
bone_bindings[0].binding = 0;
bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
bone_bindings[0].descriptorCount = 1;
bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT;
bone_bindings[1].binding = 1;
bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC;
bone_bindings[1].descriptorCount = 1;
bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT;
VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO};
bone_layout_info.bindingCount = 2;
bone_layout_info.pBindings = bone_bindings;
check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone");
VkDescriptorPoolSize pool_sizes[3] = {
{VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384},
{VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4},
{VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4}};
VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO};
pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT;
pool_info.maxSets = 8192;
pool_info.poolSizeCount = 3;
pool_info.pPoolSizes = pool_sizes;
check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool");
const std::uint8_t white_pixel[4] = {255, 255, 255, 255};
upload_texture(1, 1, white_pixel, fallback_texture_, false, false);
fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view);
fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view);
void* bone_raw = nullptr;
create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, bone_buffer_, bone_memory_, &bone_raw);
bone_mapped_ = static_cast<float*>(bone_raw);
bone_capacity_ = kMaxBonesPerFrame;
for (std::size_t i = 0; i < 16; ++i) bone_mapped_[i] = kIdentityMatrix[i];
VkPhysicalDeviceProperties properties{};
vkGetPhysicalDeviceProperties(physical_, &properties);
const auto alignment = std::max<VkDeviceSize>(
1, properties.limits.minStorageBufferOffsetAlignment);
state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment;
state_capacity_ = 256;
void* state_raw = nullptr;
create_host_buffer(state_capacity_ * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
state_buffer_, state_memory_, &state_raw);
state_mapped_ = static_cast<std::uint8_t*>(state_raw);
VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO};
bone_alloc.descriptorPool = descriptor_pool_;
bone_alloc.descriptorSetCount = 1;
bone_alloc.pSetLayouts = &bone_descriptor_layout_;
check(vkAllocateDescriptorSets(device_, &bone_alloc, &bone_descriptor_set_), "vkAllocateDescriptorSets bone");
VkDescriptorBufferInfo buffer_infos[2] = {
{bone_buffer_, 0, kBoneBufferBytes},
{state_buffer_, 0, sizeof(FixedFunctionState)}};
VkWriteDescriptorSet writes[2]{};
for (int binding = 0; binding < 2; ++binding) {
writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
writes[binding].dstSet = bone_descriptor_set_;
writes[binding].dstBinding = static_cast<std::uint32_t>(binding);
writes[binding].descriptorCount = 1;
writes[binding].descriptorType = binding == 0
? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC;
writes[binding].pBufferInfo = &buffer_infos[binding];
}
vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr);
void* ui_raw = nullptr;
create_host_buffer(
kUiBufferBytes,
VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT,
ui_buffer_,
ui_memory_,
&ui_raw);
ui_mapped_ = static_cast<std::uint8_t*>(ui_raw);
ui_vertex_capacity_ = kMaxUiVerticesPerFrame;
ui_index_capacity_ = kMaxUiIndicesPerFrame;
ui_vertex_bytes_ = kUiVertexBytes;
}
void ensure_bone_capacity(std::size_t required) {
if (required <= bone_capacity_) return;
VkPhysicalDeviceProperties properties{};
vkGetPhysicalDeviceProperties(physical_, &properties);
const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float));
if (required > max_bones || required > UINT32_MAX)
throw std::runtime_error("bone palette exceeds device storage-buffer limit");
std::size_t next = std::min(max_bones, std::max(required, bone_capacity_ * 2));
VkBuffer buffer = VK_NULL_HANDLE;
VkDeviceMemory memory = VK_NULL_HANDLE;
void* mapped = nullptr;
create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped);
vkUnmapMemory(device_, bone_memory_);
vkDestroyBuffer(device_, bone_buffer_, nullptr);
vkFreeMemory(device_, bone_memory_, nullptr);
bone_buffer_ = buffer;
bone_memory_ = memory;
bone_mapped_ = static_cast<float*>(mapped);
bone_capacity_ = next;
VkDescriptorBufferInfo info{bone_buffer_, 0, next * 16 * sizeof(float)};
VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET};
write.dstSet = bone_descriptor_set_;
write.dstBinding = 0;
write.descriptorCount = 1;
write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
write.pBufferInfo = &info;
vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr);
}
void ensure_state_capacity(std::size_t required) {
if (required <= state_capacity_) return;
const auto next = std::max(required, state_capacity_ * 2);
if (next > UINT32_MAX / state_stride_)
throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range");
VkBuffer buffer = VK_NULL_HANDLE;
VkDeviceMemory memory = VK_NULL_HANDLE;
void* mapped = nullptr;
create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
buffer, memory, &mapped);
vkUnmapMemory(device_, state_memory_);
vkDestroyBuffer(device_, state_buffer_, nullptr);
vkFreeMemory(device_, state_memory_, nullptr);
state_buffer_ = buffer;
state_memory_ = memory;
state_mapped_ = static_cast<std::uint8_t*>(mapped);
state_capacity_ = next;
VkDescriptorBufferInfo info{state_buffer_, 0, sizeof(FixedFunctionState)};
VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET};
write.dstSet = bone_descriptor_set_;
write.dstBinding = 1;
write.descriptorCount = 1;
write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC;
write.pBufferInfo = &info;
vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr);
}
void ensure_ui_capacity(std::size_t command_count) {
if (command_count <= ui_vertex_capacity_ / 4 && command_count <= ui_index_capacity_ / 6) return;
if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit");
const std::size_t quads = std::max(command_count, ui_vertex_capacity_ / 2);
const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex);
const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t);
if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull)
throw std::runtime_error("UI geometry exceeds 128 MiB safety limit");
VkBuffer buffer = VK_NULL_HANDLE;
VkDeviceMemory memory = VK_NULL_HANDLE;
void* mapped = nullptr;
create_host_buffer(vertex_bytes + index_bytes,
VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped);
vkUnmapMemory(device_, ui_memory_);
vkDestroyBuffer(device_, ui_buffer_, nullptr);
vkFreeMemory(device_, ui_memory_, nullptr);
ui_buffer_ = buffer;
ui_memory_ = memory;
ui_mapped_ = static_cast<std::uint8_t*>(mapped);
ui_vertex_capacity_ = quads * 4;
ui_index_capacity_ = quads * 6;
ui_vertex_bytes_ = vertex_bytes;
}
void create_render_pass_and_layout() {
VkAttachmentDescription attachments[2]{};
attachments[0].format = swapchain_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT;
attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE;
attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[0].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT;
attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL};
VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL};
VkSubpassDescription subpass{};
subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref;
subpass.pDepthStencilAttachment = &depth_ref;
VkSubpassDependency dependency{};
dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0;
dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT;
dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT;
dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO};
pass_info.attachmentCount = 2; pass_info.pAttachments = attachments;
pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass;
pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency;
check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass");
for (auto view : image_views_) {
VkImageView fb_attachments[2] = {view, depth_view_};
VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO};
framebuffer.renderPass = pass_; framebuffer.attachmentCount = 2; framebuffer.pAttachments = fb_attachments;
framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1;
VkFramebuffer handle = VK_NULL_HANDLE;
check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer");
framebuffers_.push_back(handle);
}
const auto vert = read_spirv("native.vert.spv");
const auto frag = read_spirv("native.frag.spv");
auto module = [&](const std::vector<std::uint32_t>& code) {
VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO};
info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data();
VkShaderModule handle = VK_NULL_HANDLE;
check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule");
return handle;
};
vertex_module_ = module(vert);
fragment_module_ = module(frag);
VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_};
VkPushConstantRange push_constants{};
push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT;
push_constants.size = sizeof(PushConstants);
VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO};
layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts;
layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants;
check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout");
get_pipeline(0, 2, 0);
}
VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend,
bool lines = false, std::uint8_t z_func = 4) {
const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) |
(std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) |
(std::uint32_t(z_func & 0xfu) << 25);
if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second;
VkPipelineShaderStageCreateInfo stages[2]{};
stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main";
stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main";
VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX};
VkVertexInputAttributeDescription attributes[9] = {
{0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)},
{1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)},
{2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)},
{3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)},
{4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)},
{5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)},
{6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)},
{7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)},
{8, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, vertex_fog)}};
VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO};
input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding;
input.vertexAttributeDescriptionCount = 9; input.pVertexAttributeDescriptions = attributes;
VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO};
assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST;
VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1};
VkRect2D scissor{{0, 0}, extent_};
VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO};
viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport;
viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor;
// D3DVIEWPORT8 changes per draw (SetViewport), so the viewport is dynamic state.
const VkDynamicState dynamic_states[] = {VK_DYNAMIC_STATE_VIEWPORT};
VkPipelineDynamicStateCreateInfo dynamic_info{VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO};
dynamic_info.dynamicStateCount = 1; dynamic_info.pDynamicStates = dynamic_states;
VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO};
raster.polygonMode = VK_POLYGON_MODE_FILL;
raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE);
raster.frontFace = VK_FRONT_FACE_CLOCKWISE;
raster.lineWidth = 1;
VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO};
multisample.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT;
VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO};
depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE;
depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE;
switch (z_func) {
case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break;
case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break;
case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break;
case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break;
case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break;
case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break;
case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break;
default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break;
}
auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor {
switch (d3d) {
case 1: return VK_BLEND_FACTOR_ZERO;
case 2: return VK_BLEND_FACTOR_ONE;
case 3: return VK_BLEND_FACTOR_SRC_COLOR;
case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR;
case 5: return VK_BLEND_FACTOR_SRC_ALPHA;
case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA;
case 7: return VK_BLEND_FACTOR_DST_ALPHA;
case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA;
case 9: return VK_BLEND_FACTOR_DST_COLOR;
case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR;
case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE;
default: return fallback;
}
};
VkPipelineColorBlendAttachmentState blend_attachment{};
blend_attachment.colorWriteMask = 0xf;
if (blend > 0) {
blend_attachment.blendEnable = VK_TRUE;
if (blend >= 16) {
blend_attachment.srcColorBlendFactor =
d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA);
blend_attachment.dstColorBlendFactor =
d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA);
} else {
blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA;
blend_attachment.dstColorBlendFactor =
blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA;
}
blend_attachment.colorBlendOp = VK_BLEND_OP_ADD;
// D3D8 has no separate alpha blend: SRCBLEND/DESTBLEND apply to alpha as well.
auto alpha_factor = [](VkBlendFactor factor) {
switch (factor) {
case VK_BLEND_FACTOR_SRC_COLOR: return VK_BLEND_FACTOR_SRC_ALPHA;
case VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA;
case VK_BLEND_FACTOR_DST_COLOR: return VK_BLEND_FACTOR_DST_ALPHA;
case VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA;
case VK_BLEND_FACTOR_SRC_ALPHA_SATURATE: return VK_BLEND_FACTOR_ONE;
default: return factor;
}
};
blend_attachment.srcAlphaBlendFactor = alpha_factor(blend_attachment.srcColorBlendFactor);
blend_attachment.dstAlphaBlendFactor = alpha_factor(blend_attachment.dstColorBlendFactor);
blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD;
}
VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO};
blending.attachmentCount = 1; blending.pAttachments = &blend_attachment;
VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO};
pipeline_info.stageCount = 2; pipeline_info.pStages = stages;
pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly;
pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster;
pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil;
pipeline_info.pColorBlendState = &blending;
pipeline_info.pDynamicState = &dynamic_info;
pipeline_info.layout = layout_; pipeline_info.renderPass = pass_;
VkPipeline handle = VK_NULL_HANDLE;
check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines");
pipelines_.emplace(key, handle);
return handle;
}
VkDescriptorSet allocate_texture_descriptor_set(VkImageView view0, VkImageView view1,
VkSampler sampler0 = VK_NULL_HANDLE,
VkSampler sampler1 = VK_NULL_HANDLE) {
VkDescriptorSet set = VK_NULL_HANDLE;
VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO};
desc_alloc.descriptorPool = descriptor_pool_;
desc_alloc.descriptorSetCount = 1;
desc_alloc.pSetLayouts = &descriptor_layout_;
check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets");
VkDescriptorImageInfo image_infos[2]{};
image_infos[0].sampler = sampler0 ? sampler0 : sampler_;
image_infos[0].imageView = view0;
image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_);
image_infos[1].imageView = view1;
image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
VkWriteDescriptorSet writes[2]{};
for (int b = 0; b < 2; ++b) {
writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
writes[b].dstSet = set;
writes[b].dstBinding = static_cast<std::uint32_t>(b);
writes[b].descriptorCount = 1;
writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
writes[b].pImageInfo = &image_infos[b];
}
vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr);
return set;
}
VkDescriptorSet allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) {
VkDescriptorSet set = VK_NULL_HANDLE;
VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO};
desc_alloc.descriptorPool = descriptor_pool_;
desc_alloc.descriptorSetCount = 1;
desc_alloc.pSetLayouts = &descriptor_layout_;
check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui");
VkDescriptorImageInfo image_infos[2]{};
image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_;
image_infos[0].imageView = view0;
image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_);
image_infos[1].imageView = view1;
image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
VkWriteDescriptorSet writes[2]{};
for (int b = 0; b < 2; ++b) {
writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
writes[b].dstSet = set;
writes[b].dstBinding = static_cast<std::uint32_t>(b);
writes[b].descriptorCount = 1;
writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
writes[b].pImageInfo = &image_infos[b];
}
vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr);
return set;
}
GpuTexture& get_gpu_texture(
const std::string& texture_name,
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures) {
if (texture_name.empty()) return fallback_texture_;
auto [it, inserted] = textures_.try_emplace(texture_name);
it->second.last_used_frame = frame_number_;
if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60))
return it->second;
if (inserted) {
it->second.descriptor = fallback_texture_.descriptor;
it->second.ui_descriptor = fallback_texture_.ui_descriptor;
}
it->second.last_decode_attempt_frame = frame_number_;
mtgodot::Image decoded;
if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) {
decoded = decode_texture_bytes(raw->second.data(), raw->second.size());
}
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
if (!decoded.ok()) {
if (texture_name.rfind("mem:", 0) == 0) {
UIMemoryTexture mem_tex;
if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) {
const auto mtra = native_draw_capture::encode_raw_argb_as_mtra(
static_cast<std::uint32_t>(mem_tex.width),
static_cast<std::uint32_t>(mem_tex.height),
mem_tex.argb.data());
decoded = decode_texture_bytes(mtra.data(), mtra.size());
}
} else {
std::vector<std::uint8_t> pack_bytes;
if (read_live_pack_texture(texture_name, pack_bytes))
decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size());
}
}
#endif
if (!decoded.ok()) return it->second;
bool is_ui_tex = false;
if (texture_name.rfind("icon/", 0) == 0 ||
texture_name.rfind("d:/ymir work/ui/", 0) == 0 ||
texture_name.rfind("d:\\ymir work\\ui\\", 0) == 0 ||
texture_name.rfind("locale/", 0) == 0 ||
texture_name.rfind("locale\\", 0) == 0 ||
texture_name.rfind("mem:", 0) == 0 ||
(decoded.w <= 128 && decoded.h <= 128)) {
is_ui_tex = true;
}
upload_texture(decoded.w, decoded.h, decoded.rgba.data(), it->second, true, !is_ui_tex);
it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view);
it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view);
return it->second;
}
std::uint64_t sampler_state_key(const Render3DDraw& draw, int stage) const {
return (std::uint64_t(draw.address_u[stage] & 15u)) |
(std::uint64_t(draw.address_v[stage] & 15u) << 4) |
(std::uint64_t(draw.min_filter[stage] & 15u) << 8) |
(std::uint64_t(draw.mag_filter[stage] & 15u) << 12) |
(std::uint64_t(draw.mip_filter[stage] & 15u) << 16);
}
VkSampler sampler_for_draw(const Render3DDraw& draw, int stage) {
const auto key = sampler_state_key(draw, stage);
if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_;
if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second;
auto address_mode = [](std::uint32_t mode) {
switch (mode) {
case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT;
case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER;
default: return VK_SAMPLER_ADDRESS_MODE_REPEAT;
}
};
VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO};
info.addressModeU = address_mode(draw.address_u[stage]);
info.addressModeV = address_mode(draw.address_v[stage]);
info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT;
info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR;
info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR;
info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR;
info.minLod = 0.0f;
info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE;
if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) {
info.anisotropyEnable = VK_TRUE;
info.maxAnisotropy = max_anisotropy_;
}
VkSampler sampler = VK_NULL_HANDLE;
check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state");
state_samplers_.emplace(key, sampler);
return sampler;
}
VkDescriptorSet get_texture_descriptor(
const std::string& texture0_name,
const std::string& mask_name,
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures,
const Render3DDraw& draw) {
const auto& tex0 = get_gpu_texture(texture0_name, capture_textures);
const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures);
const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" +
std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1));
if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) {
it->second.last_used_frame = frame_number_;
return it->second.set;
}
const VkDescriptorSet set = allocate_texture_descriptor_set(
tex0.view ? tex0.view : fallback_texture_.view,
tex1.view ? tex1.view : fallback_texture_.view,
sampler_for_draw(draw, 0), sampler_for_draw(draw, 1));
paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_});
return set;
}
VkDescriptorSet get_ui_texture_descriptor(
const std::string& texture0_name,
const std::string& mask_name,
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures) {
const auto& tex0 = get_gpu_texture(texture0_name, capture_textures);
if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor;
const auto& tex1 = get_gpu_texture(mask_name, capture_textures);
if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor;
const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name;
if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) {
it->second.last_used_frame = frame_number_;
return it->second.set;
}
const VkDescriptorSet set = allocate_ui_texture_descriptor_set(
tex0.view ? tex0.view : fallback_texture_.view,
tex1.view ? tex1.view : fallback_texture_.view);
paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_});
return set;
}
void upload_texture(
std::uint32_t width,
std::uint32_t height,
const std::uint8_t* rgba,
GpuTexture& texture,
bool count_upload,
bool generate_mips = true) {
const VkDeviceSize byte_size = VkDeviceSize(width) * height * 4;
const std::uint32_t mip_levels = generate_mips
? (static_cast<std::uint32_t>(std::floor(std::log2(std::max(width, height)))) + 1u)
: 1u;
VkBuffer staging_buffer = VK_NULL_HANDLE;
VkDeviceMemory staging_memory = VK_NULL_HANDLE;
VkBufferCreateInfo buffer_info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
buffer_info.size = byte_size;
buffer_info.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT;
buffer_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
check(vkCreateBuffer(device_, &buffer_info, nullptr, &staging_buffer), "vkCreateBuffer staging");
VkMemoryRequirements buffer_reqs{};
vkGetBufferMemoryRequirements(device_, staging_buffer, &buffer_reqs);
VkMemoryAllocateInfo buffer_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
buffer_alloc.allocationSize = buffer_reqs.size;
buffer_alloc.memoryTypeIndex = find_memory_type(
buffer_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
check(vkAllocateMemory(device_, &buffer_alloc, nullptr, &staging_memory), "vkAllocateMemory staging");
check(vkBindBufferMemory(device_, staging_buffer, staging_memory, 0), "vkBindBufferMemory staging");
void* mapped = nullptr;
check(vkMapMemory(device_, staging_memory, 0, byte_size, 0, &mapped), "vkMapMemory staging");
std::memcpy(mapped, rgba, byte_size);
vkUnmapMemory(device_, staging_memory);
VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
image_info.imageType = VK_IMAGE_TYPE_2D;
image_info.format = VK_FORMAT_R8G8B8A8_UNORM;
image_info.extent = {width, height, 1};
image_info.mipLevels = mip_levels;
image_info.arrayLayers = 1;
image_info.samples = VK_SAMPLE_COUNT_1_BIT;
image_info.tiling = VK_IMAGE_TILING_OPTIMAL;
image_info.usage =
VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT;
image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture");
VkMemoryRequirements image_reqs{};
vkGetImageMemoryRequirements(device_, texture.image, &image_reqs);
VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
image_alloc.allocationSize = image_reqs.size;
image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture");
check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture");
VkCommandBuffer upload_cmd = VK_NULL_HANDLE;
VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1;
check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture");
VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture");
VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED;
to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
to_transfer.image = texture.image;
to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1};
vkCmdPipelineBarrier(
upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT,
0, 0, nullptr, 0, nullptr, 1, &to_transfer);
VkBufferImageCopy region{};
region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1};
region.imageExtent = {width, height, 1};
vkCmdCopyBufferToImage(upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &region);
std::int32_t mip_w = static_cast<std::int32_t>(width);
std::int32_t mip_h = static_cast<std::int32_t>(height);
for (std::uint32_t i = 1; i < mip_levels; ++i) {
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
barrier.image = texture.image;
barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1};
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
vkCmdPipelineBarrier(
upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT,
0, 0, nullptr, 0, nullptr, 1, &barrier);
const std::int32_t next_w = std::max(1, mip_w / 2);
const std::int32_t next_h = std::max(1, mip_h / 2);
VkImageBlit blit{};
blit.srcOffsets[0] = {0, 0, 0};
blit.srcOffsets[1] = {mip_w, mip_h, 1};
blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1};
blit.dstOffsets[0] = {0, 0, 0};
blit.dstOffsets[1] = {next_w, next_h, 1};
blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1};
vkCmdBlitImage(
upload_cmd,
texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
1, &blit, VK_FILTER_LINEAR);
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
vkCmdPipelineBarrier(
upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT,
0, 0, nullptr, 0, nullptr, 1, &barrier);
mip_w = next_w;
mip_h = next_h;
}
VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
to_shader.image = texture.image;
to_shader.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1};
vkCmdPipelineBarrier(
upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT,
0, 0, nullptr, 0, nullptr, 1, &to_shader);
check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture");
VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO};
submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd;
check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture");
check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture");
vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd);
vkDestroyBuffer(device_, staging_buffer, nullptr);
vkFreeMemory(device_, staging_memory, nullptr);
VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D;
view_info.format = VK_FORMAT_R8G8B8A8_UNORM;
view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1};
check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture");
if (count_upload) {
++texture_upload_count_;
texture_uploaded_bytes_ += byte_size;
}
}
void release_texture(GpuTexture& texture) {
for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor})
if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor)
vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set);
if (texture.view) vkDestroyImageView(device_, texture.view, nullptr);
if (texture.image) vkDestroyImage(device_, texture.image, nullptr);
if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr);
texture = {};
}
void prune_textures() {
constexpr std::uint64_t static_retention_frames = 120;
constexpr std::uint64_t memory_retention_frames = 2;
for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) {
const auto retention = it->first.find("mem:") != std::string::npos
? memory_retention_frames : static_retention_frames;
if (frame_number_ - it->second.last_used_frame > retention) {
vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set);
it = paired_descriptors_.erase(it);
} else ++it;
}
for (auto it = textures_.begin(); it != textures_.end();) {
const auto retention = it->first.rfind("mem:", 0) == 0
? memory_retention_frames : static_retention_frames;
if (frame_number_ - it->second.last_used_frame > retention) {
release_texture(it->second);
it = textures_.erase(it);
} else ++it;
}
}
void release_geometry(Geometry& geometry) {
if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr);
if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr);
geometry = {};
}
void upload_geometry(const Render3DDraw& draw, Geometry& geometry) {
if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3))
throw std::runtime_error("invalid native geometry array size");
const auto count = draw.positions.size() / 3;
if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX)
throw std::runtime_error("geometry exceeds Vulkan index range");
// D3D8 feeds opaque white for a vertex format without D3DFVF_DIFFUSE.
const std::uint32_t default_argb = 0xffffffffu;
const bool has_skin =
draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4;
std::vector<Vertex> vertices(count);
for (std::size_t i = 0; i < count; ++i) {
vertices[i].position[0] = draw.positions[3*i];
vertices[i].position[1] = draw.positions[3*i+1];
vertices[i].position[2] = draw.positions[3*i+2];
vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f;
vertices[i].vertex_fog = i < draw.vertex_fog.size() ? draw.vertex_fog[i] : 1.0f;
if (3*i + 2 < draw.normals.size()) {
vertices[i].normal[0] = draw.normals[3*i];
vertices[i].normal[1] = draw.normals[3*i+1];
vertices[i].normal[2] = draw.normals[3*i+2];
} else {
vertices[i].normal[0] = 0.0f;
vertices[i].normal[1] = 0.0f;
vertices[i].normal[2] = 1.0f;
}
if (2*i + 1 < draw.uv0.size()) {
vertices[i].uv[0] = draw.uv0[2*i];
vertices[i].uv[1] = draw.uv0[2*i+1];
} else {
vertices[i].uv[0] = 0.0f;
vertices[i].uv[1] = 0.0f;
}
const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb;
vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f;
vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f;
vertices[i].color[2] = float(argb & 255) / 255.0f;
vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f;
if (has_skin) {
for (int k = 0; k < 4; ++k) {
vertices[i].joints[k] = draw.bone_indices[4*i + k];
vertices[i].weights[k] = draw.bone_weights[4*i + k];
}
} else {
vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0;
vertices[i].weights[0] = 1.0f;
vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f;
}
if (2*i + 1 < draw.uv1.size()) {
vertices[i].mask_uv[0] = draw.uv1[2*i];
vertices[i].mask_uv[1] = draw.uv1[2*i+1];
} else {
vertices[i].mask_uv[0] = 0.0f;
vertices[i].mask_uv[1] = 0.0f;
}
}
for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count");
geometry.index_offset = vertices.size() * sizeof(Vertex);
const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t);
if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB");
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT;
info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer");
VkMemoryRequirements requirements{};
vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements);
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
allocation.allocationSize = requirements.size;
allocation.memoryTypeIndex = find_memory_type(
requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory");
check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory");
void* mapped = nullptr;
check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory");
std::memcpy(mapped, vertices.data(), geometry.index_offset);
std::memcpy(static_cast<std::uint8_t*>(mapped) + geometry.index_offset,
draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t));
vkUnmapMemory(device_, geometry.memory);
geometry.vertex_count = static_cast<std::uint32_t>(count);
geometry.index_count = static_cast<std::uint32_t>(draw.indices.size());
++upload_count_;
uploaded_bytes_ += bytes;
}
bool vsync_ = true;
VkPresentModeKHR present_mode_ = VK_PRESENT_MODE_FIFO_KHR;
float timestamp_period_ns_ = 1.0f;
float max_anisotropy_ = 1.0f;
SDL_Window* window_ = nullptr;
VkInstance instance_ = VK_NULL_HANDLE;
VkSurfaceKHR surface_ = VK_NULL_HANDLE;
VkPhysicalDevice physical_ = VK_NULL_HANDLE;
VkDevice device_ = VK_NULL_HANDLE;
VkQueue queue_ = VK_NULL_HANDLE;
std::uint32_t queue_family_ = 0;
std::string device_name_;
VkSwapchainKHR swapchain_ = VK_NULL_HANDLE;
VkFormat swapchain_format_ = VK_FORMAT_UNDEFINED;
VkExtent2D extent_{};
std::vector<VkImageView> image_views_;
std::vector<VkImage> swapchain_images_;
// --screenshot-out: read back the swapchain image of one frame into a BMP file.
std::string screenshot_path_;
std::uint64_t screenshot_frame_ = 0;
VkImage depth_image_ = VK_NULL_HANDLE;
VkDeviceMemory depth_memory_ = VK_NULL_HANDLE;
VkImageView depth_view_ = VK_NULL_HANDLE;
VkRenderPass pass_ = VK_NULL_HANDLE;
std::vector<VkFramebuffer> framebuffers_;
VkSampler sampler_ = VK_NULL_HANDLE;
VkSampler clamp_sampler_ = VK_NULL_HANDLE;
VkSampler ui_sampler_ = VK_NULL_HANDLE;
std::unordered_map<std::uint64_t, VkSampler> state_samplers_;
VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE;
VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE;
VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE;
VkDescriptorSet bone_descriptor_set_ = VK_NULL_HANDLE;
VkBuffer bone_buffer_ = VK_NULL_HANDLE;
VkDeviceMemory bone_memory_ = VK_NULL_HANDLE;
float* bone_mapped_ = nullptr;
std::size_t bone_capacity_ = 0;
VkBuffer state_buffer_ = VK_NULL_HANDLE;
VkDeviceMemory state_memory_ = VK_NULL_HANDLE;
std::uint8_t* state_mapped_ = nullptr;
VkDeviceSize state_stride_ = 0;
std::size_t state_capacity_ = 0;
VkBuffer ui_buffer_ = VK_NULL_HANDLE;
VkDeviceMemory ui_memory_ = VK_NULL_HANDLE;
std::uint8_t* ui_mapped_ = nullptr;
std::size_t ui_vertex_capacity_ = 0, ui_index_capacity_ = 0;
VkDeviceSize ui_vertex_bytes_ = 0;
GpuTexture fallback_texture_{};
std::unordered_map<std::string, GpuTexture> textures_;
std::unordered_map<std::string, PairedDescriptor> paired_descriptors_;
VkShaderModule vertex_module_ = VK_NULL_HANDLE;
VkShaderModule fragment_module_ = VK_NULL_HANDLE;
VkPipelineLayout layout_ = VK_NULL_HANDLE;
std::unordered_map<std::uint32_t, VkPipeline> pipelines_;
std::unordered_map<GeometryId, Geometry, GeometryIdHash> geometries_;
std::uint64_t frame_number_ = 0;
std::uint64_t timed_frames_ = 0;
std::size_t upload_count_ = 0, uploaded_bytes_ = 0;
std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0;
VkCommandPool pool_ = VK_NULL_HANDLE;
VkCommandBuffer command_ = VK_NULL_HANDLE;
VkQueryPool query_pool_ = VK_NULL_HANDLE;
bool has_pending_query_ = false;
bool os_cursor_hidden_ = false;
bool hardware_cursor_enabled_ = false;
std::array<SDL_Cursor*, 15> game_cursors_{};
SDL_Cursor* fallback_cursor_ = nullptr;
int current_cursor_shape_ = -1;
int last_mx_ = 0, last_my_ = 0;
VkSemaphore acquire_ = VK_NULL_HANDLE, rendered_ = VK_NULL_HANDLE;
VkFence fence_ = VK_NULL_HANDLE;
std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0;
std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0;
std::size_t last_vertex_count_ = 0, last_index_count_ = 0;
Timings timings_{};
};
std::vector<Render3DDraw> make_test_draws(std::size_t count, std::size_t triangles_per_draw) {
std::vector<Render3DDraw> draws;
draws.reserve(count);
for (std::size_t i = 0; i < count; ++i) {
Render3DDraw draw{};
draw.geometry_key = i + 1;
draw.geometry_revision = 1;
draw.z_enable = 1;
draw.z_write = 1;
for (auto* matrix : {draw.world, draw.view, draw.proj})
for (int diagonal = 0; diagonal < 4; ++diagonal) matrix[diagonal * 5] = 1.0f;
const float x = (float(i % 8) - 3.5f) * 0.24f;
const float y = (float(i / 8) - 3.5f) * 0.22f;
draw.positions.reserve(triangles_per_draw * 9);
draw.indices.reserve(triangles_per_draw * 3);
draw.diffuse.reserve(triangles_per_draw * 3);
for (std::size_t t = 0; t < triangles_per_draw; ++t) {
const float tx = x + (float(t % 16) - 7.5f) * 0.009f;
const float ty = y + (float(t / 16) - float(triangles_per_draw / 32)) * 0.006f;
const auto base = static_cast<std::uint32_t>(draw.positions.size() / 3);
draw.positions.insert(draw.positions.end(), {tx-0.004f, ty-0.004f, 0.5f,
tx+0.004f, ty-0.004f, 0.5f,
tx, ty+0.004f, 0.5f});
draw.indices.insert(draw.indices.end(), {base, base+1, base+2});
draw.diffuse.insert(draw.diffuse.end(), {0xff4fbfff, 0xff4fbfff, 0xff4fbfff});
}
draws.push_back(std::move(draw));
}
return draws;
}
void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms = 0.0,
std::vector<double> frame_ms = {}) {
const auto timing = renderer.timings();
std::sort(frame_ms.begin(), frame_ms.end());
const auto percentile = [&](double fraction) {
if (frame_ms.empty()) return 0.0;
const auto index = static_cast<std::size_t>(std::ceil(fraction * frame_ms.size())) - 1;
return frame_ms[std::min(index, frame_ms.size() - 1)];
};
std::cout << "device=" << renderer.device_name()
<< " present_mode=" << renderer.present_mode_name()
<< " frames=" << completed
<< " draws=" << renderer.draw_count()
<< " skinned_draws=" << renderer.skinned_draw_count()
<< " ui_batches=" << renderer.ui_batch_count()
<< " ui_quads=" << renderer.ui_quad_count()
<< " vertices=" << renderer.vertex_count()
<< " indices=" << renderer.index_count()
<< " uploads=" << renderer.upload_count()
<< " uploaded_bytes=" << renderer.uploaded_bytes()
<< " texture_uploads=" << renderer.texture_upload_count()
<< " texture_bytes=" << renderer.texture_uploaded_bytes()
<< " mean_frame_ms=" << (completed ? wall_ms / completed : 0.0)
<< " p95_frame_ms=" << percentile(0.95)
<< " p99_frame_ms=" << percentile(0.99)
<< " max_frame_ms=" << percentile(1.0)
<< " game_update_ms=" << (completed ? update_ms / completed : 0.0)
<< " sync_ms=" << (completed ? timing.sync_ms / completed : 0.0)
<< " prepare_ms=" << (completed ? timing.prepare_ms / completed : 0.0)
<< " steady_prepare_ms=" << (completed > 1 ? timing.steady_prepare_ms / (completed - 1) : 0.0)
<< " submit_ms=" << (completed ? timing.submit_ms / completed : 0.0)
<< " present_ms=" << (completed ? timing.present_ms / completed : 0.0)
<< " gpu_ms=" << (timing.gpu_samples ? timing.gpu_ms / double(timing.gpu_samples) : 0.0) << '\n';
}
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
bool py_exec(const std::string& code) {
std::string err;
if (!PythonBoot::RunLine(code.c_str(), &err)) {
std::cerr << "PythonBoot::RunLine failed: " << err << '\n';
return false;
}
return true;
}
bool py_eval_true(const char* expr) {
std::string res, err;
return PythonBoot::Evaluate(expr, &res, &err) && (res == "True" || res == "1");
}
bool parse_live_server_spec(const std::string& spec, std::string& host, int& auth_port, int& game_port) {
const auto p1 = spec.find(':');
if (p1 == std::string::npos) return false;
const auto p2 = spec.find(':', p1 + 1);
if (p2 == std::string::npos) return false;
host = spec.substr(0, p1);
auth_port = std::stoi(spec.substr(p1 + 1, p2 - p1 - 1));
game_port = std::stoi(spec.substr(p2 + 1));
return !host.empty() && auth_port > 0 && game_port > 0;
}
int run_live_client(
VulkanWindow& renderer,
const std::string& client_dir,
int frames,
int fake_mobs,
bool gpu_skinning,
bool native_terrain,
bool login_screen,
const std::string& live_server_spec,
const std::string& capture_out) {
if (fake_mobs > 0) {
const std::string mobs_str = std::to_string(fake_mobs);
setenv("MT_FAKE_MOB_COUNT", mobs_str.c_str(), 1);
}
char resolved_client_dir[4096] = {};
const std::string abs_client_dir = realpath(client_dir.c_str(), resolved_client_dir) ? std::string(resolved_client_dir) : client_dir;
setenv("MT_40250_CLIENT", abs_client_dir.c_str(), 1);
if (chdir(abs_client_dir.c_str()) != 0 || !mtpack40250::initialize(".")) {
throw std::runtime_error("cannot initialize 40250 pack at: " + abs_client_dir);
}
const bool host_hardware_cursor = !renderer.touch_controller.is_enabled();
PythonBoot::SetHostHardwareCursorEnabled(host_hardware_cursor);
if (host_hardware_cursor) renderer.enable_game_hardware_cursor();
SetNativeTerrainRenderEnabled(native_terrain);
SetGpuSkinningEnabled(gpu_skinning);
NativeAudioEngine audio;
std::string server_host = "127.0.0.1";
int auth_port = 0;
int game_port = 0;
const bool use_external_server = !live_server_spec.empty();
FakeLoginServer server;
if (use_external_server) {
if (!parse_live_server_spec(live_server_spec, server_host, auth_port, game_port))
throw std::runtime_error("invalid --live-server spec (expected HOST:AUTH_PORT:GAME_PORT): " + live_server_spec);
} else {
if (!server.Start()) throw std::runtime_error("FakeLoginServer failed to start on loopback");
auth_port = server.AuthPort();
game_port = server.GamePort();
}
std::string error;
const char* env_stdlib = std::getenv("MT_PYTHON_STDLIB");
std::string stdlib = (env_stdlib && *env_stdlib) ? env_stdlib : bundled_file_path("python27.zip");
#ifdef __ANDROID__
if (!env_stdlib || !*env_stdlib) {
std::size_t zip_size = 0;
void* zip = SDL_LoadFile("assets://python27.zip", &zip_size);
if (!zip) zip = SDL_LoadFile("python27.zip", &zip_size);
if (!zip) throw std::runtime_error("cannot load bundled python27.zip");
char* pref = SDL_GetPrefPath("mtgodot", "native-render");
if (!pref) { SDL_free(zip); throw std::runtime_error("cannot locate app data directory"); }
stdlib = std::string(pref) + "python27.zip";
SDL_free(pref);
std::ofstream output(stdlib, std::ios::binary | std::ios::trunc);
output.write(static_cast<const char*>(zip), static_cast<std::streamsize>(zip_size));
SDL_free(zip);
if (!output) throw std::runtime_error("cannot extract bundled python27.zip");
}
#endif
if (!PythonBoot::Start(stdlib.c_str(), &error))
throw std::runtime_error("PythonBoot::Start: " + error);
SetPlatformServerTime(123456789);
PythonBoot::SetUISize(static_cast<int>(renderer.logical_width()), static_cast<int>(renderer.logical_height()));
if (!PythonBoot::RunMainScript("", &error) || !PythonBoot::IsAppLooping())
throw std::runtime_error("PythonBoot::RunMainScript: " + error);
const std::unordered_map<std::string, std::vector<std::uint8_t>> empty_textures;
auto pump_until = [&](double max_seconds, int sleep_ms, const auto& cond) -> bool {
const auto deadline = std::chrono::steady_clock::now() + std::chrono::duration<double>(max_seconds);
while (std::chrono::steady_clock::now() < deadline && renderer.poll(true) && PythonBoot::IsAppLooping()) {
PythonBoot::UIUpdate();
renderer.sync_game_cursor();
PythonBoot::UIRender();
audio.pump();
unsigned ui_w = renderer.width(), ui_h = renderer.height();
UIRenderGetSize(&ui_w, &ui_h);
renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands());
if (cond()) return true;
if (sleep_ms > 0)
std::this_thread::sleep_for(std::chrono::milliseconds(sleep_ms));
}
return cond();
};
if (!py_exec("import __main__, app, networkModule, introLogin, introSelect, introLoading, game\n"
"__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]"))
throw std::runtime_error("failed to locate networkModule.MainStream");
if (!pump_until(20.0, 5, [] { return py_eval_true("isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)"); }))
throw std::runtime_error("timed out waiting for LoginWindow");
if (login_screen) {
char conn_cmd[512];
std::snprintf(
conn_cmd,
sizeof(conn_cmd),
"_stream.SetConnectInfo('%s', %d, '%s', %d)\n"
"_w = _stream.curPhaseWindow\n"
"_w._LoginWindow__OpenServerBoard()",
server_host.c_str(),
game_port,
server_host.c_str(),
auth_port);
if (!py_exec(conn_cmd))
throw std::runtime_error("failed to configure LoginWindow connection info");
for (int i = 0; i < 15; ++i) {
PythonBoot::UIUpdate();
renderer.sync_game_cursor();
PythonBoot::UIRender();
}
} else {
char login_cmd[512];
std::snprintf(
login_cmd,
sizeof(login_cmd),
"_stream.SetConnectInfo('%s', %d, '%s', %d)\n"
"_w = _stream.curPhaseWindow\n"
"_w._LoginWindow__OpenLoginBoard()\n"
"_w.idEditLine.SetText('%s')\n"
"_w.pwdEditLine.SetText('%s')\n"
"_w._LoginWindow__OnClickLoginButton()",
server_host.c_str(),
game_port,
server_host.c_str(),
auth_port,
FakeLoginServer::kLogin,
FakeLoginServer::kPassword);
if (!py_exec(login_cmd))
throw std::runtime_error("failed to submit login credentials");
if (!pump_until(20.0, 5, [&] {
return (use_external_server || server.Has("game1:login2")) &&
py_eval_true("isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)");
}))
throw std::runtime_error("timed out waiting for SelectCharacterWindow");
if (!py_exec("_stream.curPhaseWindow.SelectSlot(0)\n"
"_stream.curPhaseWindow.StartGame()"))
throw std::runtime_error("failed to start game from SelectCharacterWindow");
if (!use_external_server) {
if (!pump_until(30.0, 5, [&] { return server.Has("game2:client_version"); }))
throw std::runtime_error("timed out waiting for LoadingWindow (client_version)");
}
if (!pump_until(60.0, 0, [&] {
return (use_external_server || server.Has("game2:burst_pong")) &&
py_eval_true("isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()");
}))
throw std::runtime_error("timed out waiting for GameWindow (server error: " + server.Error() + ")");
const int warmup_frames = std::max(10, fake_mobs / 2 + 10);
for (int i = 0; i < warmup_frames; ++i) {
if (!renderer.poll(true) || !PythonBoot::IsAppLooping()) break;
PythonBoot::UIUpdate();
renderer.sync_game_cursor();
audio.pump();
}
}
// Render one warmup frame to upload initial scene & UI textures/geometries before timed benchmark.
{
unsigned ui_w = renderer.width(), ui_h = renderer.height();
UIRenderGetSize(&ui_w, &ui_h);
renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands());
}
if (!capture_out.empty()) {
std::unordered_map<std::string, std::vector<std::uint8_t>> cap_textures;
auto add_tex = [&](const std::string& name) {
if (name.empty() || cap_textures.count(name)) return;
if (name.rfind("mem:", 0) == 0) {
UIMemoryTexture mem_tex;
if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) {
auto mtra = native_draw_capture::encode_raw_argb_as_mtra(
static_cast<std::uint32_t>(mem_tex.width),
static_cast<std::uint32_t>(mem_tex.height),
mem_tex.argb.data());
if (!mtra.empty()) cap_textures.emplace(name, std::move(mtra));
}
return;
}
std::vector<std::uint8_t> bytes;
if (read_live_pack_texture(name, bytes))
cap_textures.emplace(name, std::move(bytes));
};
for (const auto& d : Render3DDraws()) {
add_tex(d.texture0);
add_tex(d.texture1);
}
for (const auto& c : UIRenderCommands()) {
if (c.kind == UIRenderCommand::Image) {
add_tex(c.text);
add_tex(c.mask);
}
}
unsigned ui_w = renderer.width(), ui_h = renderer.height();
UIRenderGetSize(&ui_w, &ui_h);
std::vector<UIRenderCommand> cap_ui = UIRenderCommands();
if (renderer.touch_controller.is_enabled()) {
renderer.touch_controller.update_screen_size(int(ui_w), int(ui_h));
renderer.touch_controller.append_ui_commands(cap_ui);
}
native_draw_capture::write(capture_out, Render3DDraws(), cap_textures, ui_w, ui_h, cap_ui);
}
renderer.reset_timings();
const auto start = std::chrono::steady_clock::now();
double update_ms = 0.0;
int completed = 0;
std::vector<double> frame_ms;
if (frames > 0) frame_ms.reserve(static_cast<std::size_t>(frames));
while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) {
const auto frame_start = std::chrono::steady_clock::now();
const auto t0 = std::chrono::steady_clock::now();
PythonBoot::UIUpdate();
renderer.sync_game_cursor();
PythonBoot::UIRender();
audio.pump();
const auto t1 = std::chrono::steady_clock::now();
update_ms += std::chrono::duration<double, std::milli>(t1 - t0).count();
unsigned ui_w = renderer.width(), ui_h = renderer.height();
UIRenderGetSize(&ui_w, &ui_h);
renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands());
frame_ms.push_back(std::chrono::duration<double, std::milli>(
std::chrono::steady_clock::now() - frame_start).count());
++completed;
}
renderer.finish_gpu_timings();
const auto wall_ms = std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now() - start).count();
print_summary(renderer, completed, wall_ms, update_ms, std::move(frame_ms));
PythonBoot::Stop();
if (!use_external_server)
server.Stop();
return (frames == 0 || completed == frames) ? 0 : 2;
}
#endif
} // namespace
int main(int argc, char** argv) {
try {
int frames = 180;
bool frames_specified = false;
bool synthetic_specified = false;
bool interactive = false;
bool login_screen = false;
int init_width = 1280;
int init_height = 800;
std::size_t draw_count = 64;
std::size_t triangles_per_draw = 333;
std::string capture_path;
std::string capture_out;
std::string screenshot_out;
std::string live_client_dir;
std::string live_server_spec;
int fake_mobs = 64;
bool animate_first_draw = false;
bool animate_bones = false;
bool vsync = true;
bool gpu_skinning = true;
bool native_terrain = true;
bool mobile_mode = false;
#ifdef __ANDROID__
mobile_mode = true;
#endif
for (int i = 1; i < argc; ++i) {
const std::string arg = argv[i];
if (arg == "--frames" && i+1 < argc) {
frames = std::stoi(argv[++i]);
frames_specified = true;
} else if (arg == "--interactive") interactive = true;
else if (arg == "--login-screen") {
login_screen = true;
interactive = true;
} else if (arg == "--live-server" && i+1 < argc) {
live_server_spec = argv[++i];
interactive = true;
}
else if (arg == "--width" && i+1 < argc) init_width = std::stoi(argv[++i]);
else if (arg == "--height" && i+1 < argc) init_height = std::stoi(argv[++i]);
else if (arg == "--draws" && i+1 < argc) {
draw_count = std::stoul(argv[++i]);
synthetic_specified = true;
}
else if (arg == "--triangles-per-draw" && i+1 < argc) {
triangles_per_draw = std::stoul(argv[++i]);
synthetic_specified = true;
}
else if (arg == "--capture" && i+1 < argc) {
capture_path = argv[++i];
synthetic_specified = true;
}
else if (arg == "--capture-out" && i+1 < argc) capture_out = argv[++i];
else if (arg == "--screenshot-out" && i+1 < argc) screenshot_out = argv[++i];
else if (arg == "--live-client" && i+1 < argc) live_client_dir = argv[++i];
else if (arg == "--fake-mobs" && i+1 < argc) fake_mobs = std::stoi(argv[++i]);
else if (arg == "--animate-first-draw") {
animate_first_draw = true;
synthetic_specified = true;
}
else if (arg == "--animate-bones") {
animate_bones = true;
synthetic_specified = true;
}
else if (arg == "--no-vsync") vsync = false;
else if (arg == "--gpu-skinning") gpu_skinning = true;
else if (arg == "--no-gpu-skinning") gpu_skinning = false;
else if (arg == "--no-terrain") native_terrain = false;
else if (arg == "--mobile") mobile_mode = true;
else throw std::runtime_error(
"usage: mt_native_render [--frames N] [--interactive] [--login-screen] "
"[--live-server HOST:AUTH_PORT:GAME_PORT] [--width W] [--height H] "
"[--draws N] [--triangles-per-draw N] [--capture FILE] [--animate-first-draw] "
"[--animate-bones] [--no-vsync] [--live-client DIR] [--fake-mobs N] "
"[--gpu-skinning|--no-gpu-skinning] [--no-terrain] [--mobile] [--capture-out FILE] [--screenshot-out FILE.bmp]");
}
if (live_client_dir.empty() && !synthetic_specified) {
const char* env_client = std::getenv("MT_40250_CLIENT");
if (env_client && *env_client && access((std::string(env_client) + "/pack/Index").c_str(), R_OK) == 0) {
live_client_dir = env_client;
} else if (access("Client/pack/Index", R_OK) == 0) {
live_client_dir = "Client";
} else if (access("../Client/pack/Index", R_OK) == 0) {
live_client_dir = "../Client";
} else if (access("../../Client/pack/Index", R_OK) == 0) {
live_client_dir = "../../Client";
}
#ifdef __ANDROID__
if (live_client_dir.empty()) {
char* pref = SDL_GetPrefPath("mtgodot", "native-render");
if (pref) {
const std::string candidate = std::string(pref) + "Client";
if (access((candidate + "/pack/Index").c_str(), R_OK) == 0)
live_client_dir = candidate;
SDL_free(pref);
}
}
#endif
if (!live_client_dir.empty() && !frames_specified) {
interactive = true;
}
}
if (interactive && !frames_specified) frames = 0;
if (frames < 0 || (!interactive && frames == 0) || draw_count == 0 || draw_count > 4096 ||
triangles_per_draw == 0 || triangles_per_draw > 10000)
throw std::runtime_error("invalid frame/draw/triangle count");
VulkanWindow renderer(vsync, init_width, init_height);
if (!screenshot_out.empty() && frames > 0) renderer.request_screenshot(screenshot_out, std::uint64_t(frames));
if (mobile_mode) {
renderer.touch_controller.set_enabled(true);
renderer.touch_controller.update_screen_size(init_width, init_height);
}
if (!live_client_dir.empty()) {
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
return run_live_client(
renderer,
live_client_dir,
frames,
fake_mobs,
gpu_skinning,
native_terrain,
login_screen,
live_server_spec,
capture_out);
#else
throw std::runtime_error("mt_native_render was built without port_platform (--live-client unavailable)");
#endif
}
native_draw_capture::Capture capture{};
if (capture_path.empty()) {
capture.draws = make_test_draws(draw_count, triangles_per_draw);
} else {
capture = native_draw_capture::read_capture(capture_path);
}
auto& draws = capture.draws;
if (animate_first_draw && (draws.empty() || draws.front().positions.empty()))
throw std::runtime_error("first draw has no geometry to animate");
const auto start = std::chrono::steady_clock::now();
int completed = 0;
std::vector<double> frame_ms;
if (frames > 0) frame_ms.reserve(static_cast<std::size_t>(frames));
while ((frames == 0 || completed < frames) && renderer.poll()) {
const auto frame_start = std::chrono::steady_clock::now();
if (animate_first_draw && completed > 0) {
draws.front().positions[0] += 0.0001f;
++draws.front().geometry_revision;
}
if (animate_bones && completed > 0) {
const float offset = std::sin(float(completed) * 0.1f) * 0.5f;
for (auto& d : draws) {
for (std::size_t b = 12; b < d.bone_matrices.size(); b += 16)
d.bone_matrices[b] += offset;
}
}
renderer.render(draws, capture.textures, capture.ui_width, capture.ui_height, capture.ui_commands);
frame_ms.push_back(std::chrono::duration<double, std::milli>(
std::chrono::steady_clock::now() - frame_start).count());
++completed;
}
renderer.finish_gpu_timings();
const auto ms = std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now() - start).count();
print_summary(renderer, completed, ms, 0.0, std::move(frame_ms));
return (frames == 0 || completed == frames) ? 0 : 2;
} catch (const std::exception& error) {
std::cerr << "native renderer: " << error.what() << '\n';
return 1;
}
}