- --perf-log CSV + logcat telemetry (perf_log.h, EterBase/PerfCounters.h), incl. display refresh rate / app vsync rate, thermal, CPU clocks; tools/pull_perf_logs.sh + tools/perf_summary.py - Vulkan renderer: two frames in flight, in-frame texture uploads (no vkQueueWaitIdle), --fps-cap / --refresh-rate / --frames-in-flight - Android: ANativeWindow_setFrameRate, ADPF performance hint, appCategory=game (ColorOS no longer drops to 60 Hz when idle) - Character shadow map rendered on the GPU into an offscreen render target instead of CPU rasterization + per-frame 1 MB upload; --cpu-shadow keeps the old path. Device (120 Hz, 32 mobs): 75 -> ~115 fps, script thread 8.9 -> 3.4 ms. draw_capture format v10. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
4387 lines
228 KiB
C++
4387 lines
228 KiB
C++
#include "RenderCommands3D.h"
|
|
#include "UIRenderCommands.h"
|
|
#include "draw_capture.h"
|
|
#include "dxt.h"
|
|
#include "touch_controller.h"
|
|
#include "perf_log.h"
|
|
#include "android_perf.h"
|
|
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
#include "platform/MilesLib/AudioCommands.h"
|
|
#include "platform/PackBackend.h"
|
|
#include "platform/ScriptLib/PythonBoot.h"
|
|
#include "platform/UserInterface/ServerClock.h"
|
|
#include "platform/EterBase/TraceErrorObserver.h"
|
|
#include "../tests/port_login_flow_server.h"
|
|
#include <thread>
|
|
#endif
|
|
|
|
#include <unistd.h>
|
|
|
|
#include <SDL3/SDL.h>
|
|
#include <SDL3/SDL_main.h>
|
|
#include <SDL3/SDL_vulkan.h>
|
|
#ifdef __ANDROID__
|
|
#include <jni.h>
|
|
#endif
|
|
#include <vulkan/vulkan.h>
|
|
#include "stb_image.h"
|
|
|
|
#ifdef __APPLE__
|
|
#include <AudioToolbox/AudioToolbox.h>
|
|
#include <TargetConditionals.h>
|
|
#endif
|
|
|
|
#include <algorithm>
|
|
#include <array>
|
|
#include <chrono>
|
|
#include <cctype>
|
|
#include <climits>
|
|
#include <cmath>
|
|
#include <cstdint>
|
|
#include <cstdlib>
|
|
#include <cstring>
|
|
#include <fstream>
|
|
#include <iostream>
|
|
#include <optional>
|
|
#include <sstream>
|
|
#include <stdexcept>
|
|
#include <string>
|
|
#include <unordered_map>
|
|
#include <unordered_set>
|
|
#include <vector>
|
|
|
|
namespace {
|
|
|
|
void check(VkResult result, const char* operation) {
|
|
if (result != VK_SUCCESS) throw std::runtime_error(std::string(operation) + ": VkResult " + std::to_string(result));
|
|
}
|
|
|
|
struct Vertex {
|
|
float position[3];
|
|
float normal[3];
|
|
float uv[2];
|
|
float color[4];
|
|
std::uint8_t joints[4];
|
|
float weights[4];
|
|
float mask_uv[2];
|
|
float rhw = 1.0f;
|
|
float vertex_fog = 1.0f;
|
|
};
|
|
|
|
struct PushConstants {
|
|
std::array<float, 16> mvp{};
|
|
// x: bone palette base + 1 (0 = not skinned), y: D3DFVF_XYZRHW, z: UI mask batch, w: unused.
|
|
std::array<float, 4> params{};
|
|
};
|
|
static_assert(sizeof(PushConstants) <= 128, "PushConstants must fit within Vulkan's 128-byte minimum guarantee");
|
|
|
|
// std430 payload selected with a dynamic storage-buffer offset for every draw: the D3D8
|
|
// fixed-function state in effect for the draw (texture stages, texture coordinate processing,
|
|
// lighting, material, fog, alpha test). A zero flags.x marks a 2D UI batch.
|
|
struct alignas(16) FixedFunctionState {
|
|
std::array<std::array<std::uint32_t, 4>, 2> stage_color{}; // op, arg1, arg2, D3DTSS_TEXCOORDINDEX
|
|
std::array<std::array<std::uint32_t, 4>, 2> stage_alpha{}; // op, arg1, arg2, D3DTSS_TEXTURETRANSFORMFLAGS
|
|
std::array<float, 4> fog_color{};
|
|
std::array<float, 4> fog_params{}; // start, end, density, range enabled
|
|
std::array<float, 4> texture_factor{};
|
|
std::array<float, 16> world_view{};
|
|
std::array<std::uint32_t, 4> flags{}; // fixed function, texture0, texture1, fog mode
|
|
std::array<float, 16> normal_matrix{}; // inverse transpose of world * view
|
|
std::array<std::uint32_t, 4> lighting_flags{}; // LIGHTING, COLORVERTEX, vertex has diffuse, NORMALIZENORMALS
|
|
std::array<std::uint32_t, 4> material_sources{}; // DIFFUSE, AMBIENT, EMISSIVE source, LOCALVIEWER
|
|
std::array<std::uint32_t, 4> alpha_test{}; // ALPHATESTENABLE, ALPHAFUNC, ALPHAREF, unused
|
|
std::array<float, 4> material_diffuse{};
|
|
std::array<float, 4> material_ambient{};
|
|
std::array<float, 4> material_emissive{};
|
|
std::array<float, 4> global_ambient{};
|
|
std::array<std::array<float, 16>, 2> texture_matrix{};
|
|
// Lights in camera space. position.w = D3DLIGHTTYPE (0 = disabled).
|
|
std::array<std::array<float, 4>, 8> light_position_type{};
|
|
std::array<std::array<float, 4>, 8> light_direction_range{};
|
|
std::array<std::array<float, 4>, 8> light_diffuse{};
|
|
std::array<std::array<float, 4>, 8> light_ambient{};
|
|
std::array<std::array<float, 4>, 8> light_attenuation{}; // a0, a1, a2, falloff
|
|
std::array<std::array<float, 4>, 8> light_spot{}; // cos(theta / 2), cos(phi / 2)
|
|
};
|
|
static_assert(sizeof(FixedFunctionState) == 1264);
|
|
|
|
constexpr std::array<float, 16> kIdentityMatrix = {
|
|
1.0f, 0.0f, 0.0f, 0.0f,
|
|
0.0f, 1.0f, 0.0f, 0.0f,
|
|
0.0f, 0.0f, 1.0f, 0.0f,
|
|
0.0f, 0.0f, 0.0f, 1.0f};
|
|
|
|
std::array<float, 16> multiply(const float* a, const float* b) {
|
|
std::array<float, 16> result{};
|
|
for (int row = 0; row < 4; ++row)
|
|
for (int col = 0; col < 4; ++col)
|
|
for (int k = 0; k < 4; ++k)
|
|
result[row * 4 + col] += a[row * 4 + k] * b[k * 4 + col];
|
|
return result;
|
|
}
|
|
|
|
std::array<float, 16> draw_mvp(const Render3DDraw& draw) {
|
|
const auto world_view = multiply(draw.world, draw.view);
|
|
return multiply(world_view.data(), draw.proj);
|
|
}
|
|
|
|
std::array<float, 4> unpack_argb(std::uint32_t argb) {
|
|
return {
|
|
float((argb >> 16) & 255) / 255.0f,
|
|
float((argb >> 8) & 255) / 255.0f,
|
|
float(argb & 255) / 255.0f,
|
|
float((argb >> 24) & 255) / 255.0f};
|
|
}
|
|
|
|
// Inverse transpose of the upper 3x3 of a row-vector matrix, the D3D8 normal transform.
|
|
std::array<float, 16> normal_matrix(const std::array<float, 16>& m) {
|
|
const float a = m[0], b = m[1], c = m[2], d = m[4], e = m[5], f = m[6], g = m[8], h = m[9], i = m[10];
|
|
const float A = e * i - f * h, B = -(d * i - f * g), C = d * h - e * g;
|
|
const float D = -(b * i - c * h), E = a * i - c * g, F = -(a * h - b * g);
|
|
const float G = b * f - c * e, H = -(a * f - c * d), I = a * e - b * d;
|
|
const float det = a * A + b * B + c * C;
|
|
std::array<float, 16> out = kIdentityMatrix;
|
|
if (det == 0.0f) return out;
|
|
const float s = 1.0f / det;
|
|
// inverse = adjugate / det; its transpose is the cofactor matrix / det.
|
|
out[0] = A * s; out[1] = B * s; out[2] = C * s;
|
|
out[4] = D * s; out[5] = E * s; out[6] = F * s;
|
|
out[8] = G * s; out[9] = H * s; out[10] = I * s;
|
|
return out;
|
|
}
|
|
|
|
PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded) {
|
|
PushConstants constants{};
|
|
constants.mvp = draw_mvp(draw);
|
|
constants.params = {skin_offset_encoded, draw.pretransformed ? 1.0f : 0.0f, 0.0f, 0.0f};
|
|
return constants;
|
|
}
|
|
|
|
FixedFunctionState make_fixed_function_state(const Render3DDraw& draw) {
|
|
FixedFunctionState state{};
|
|
for (int stage = 0; stage < 2; ++stage) {
|
|
state.stage_color[stage] = {draw.color_op[stage], draw.color_arg1[stage], draw.color_arg2[stage],
|
|
draw.texcoord_index[stage]};
|
|
state.stage_alpha[stage] = {draw.alpha_op[stage], draw.alpha_arg1[stage], draw.alpha_arg2[stage],
|
|
draw.texture_transform_flags[stage]};
|
|
std::copy(draw.texture_matrix[stage], draw.texture_matrix[stage] + 16, state.texture_matrix[stage].begin());
|
|
}
|
|
state.texture_factor = unpack_argb(draw.texture_factor);
|
|
state.world_view = multiply(draw.world, draw.view);
|
|
state.normal_matrix = normal_matrix(state.world_view);
|
|
state.fog_color = unpack_argb(draw.fog_color);
|
|
state.fog_params = {draw.fog_start, draw.fog_end, draw.fog_density,
|
|
draw.fog_range_enable ? 1.0f : 0.0f};
|
|
const std::uint32_t fog_mode = draw.fog_enable
|
|
? (draw.fog_table_mode ? draw.fog_table_mode : (draw.pretransformed ? 4u : draw.fog_vertex_mode)) : 0;
|
|
state.flags = {1u, !draw.texture0.empty() ? 1u : 0u, !draw.texture1.empty() ? 1u : 0u, fog_mode};
|
|
state.lighting_flags = {draw.lighting, draw.color_vertex, draw.diffuse.empty() ? 0u : 1u,
|
|
draw.normalize_normals};
|
|
state.material_sources = {draw.diffuse_material_source, draw.ambient_material_source,
|
|
draw.emissive_material_source, draw.local_viewer};
|
|
state.alpha_test = {draw.alpha_test, draw.alpha_func, draw.alpha_ref, 0};
|
|
std::copy(draw.material_diffuse, draw.material_diffuse + 4, state.material_diffuse.begin());
|
|
std::copy(draw.material_ambient, draw.material_ambient + 4, state.material_ambient.begin());
|
|
std::copy(draw.material_emissive, draw.material_emissive + 4, state.material_emissive.begin());
|
|
state.global_ambient = unpack_argb(draw.ambient);
|
|
const float* v = draw.view;
|
|
for (int i = 0; i < 8; ++i) {
|
|
const auto& light = draw.lights[i];
|
|
const float* p = light.position;
|
|
const float* d = light.direction;
|
|
std::array<float, 3> position{}, direction{};
|
|
for (int c = 0; c < 3; ++c) {
|
|
position[c] = p[0] * v[c] + p[1] * v[4 + c] + p[2] * v[8 + c] + v[12 + c];
|
|
direction[c] = d[0] * v[c] + d[1] * v[4 + c] + d[2] * v[8 + c];
|
|
}
|
|
const float length = std::sqrt(direction[0] * direction[0] + direction[1] * direction[1] +
|
|
direction[2] * direction[2]);
|
|
if (length > 0.0f)
|
|
for (float& c : direction) c /= length;
|
|
state.light_position_type[i] = {position[0], position[1], position[2], float(light.type)};
|
|
state.light_direction_range[i] = {direction[0], direction[1], direction[2], light.range};
|
|
std::copy(light.diffuse, light.diffuse + 4, state.light_diffuse[i].begin());
|
|
std::copy(light.ambient, light.ambient + 4, state.light_ambient[i].begin());
|
|
state.light_attenuation[i] = {light.attenuation[0], light.attenuation[1], light.attenuation[2], light.falloff};
|
|
state.light_spot[i] = {std::cos(light.theta * 0.5f), std::cos(light.phi * 0.5f), 0, 0};
|
|
}
|
|
return state;
|
|
}
|
|
|
|
std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size) {
|
|
const auto* data = static_cast<const std::uint8_t*>(bytes);
|
|
std::size_t i = 0;
|
|
for (; i + sizeof(std::uint64_t) <= size; i += sizeof(std::uint64_t)) {
|
|
std::uint64_t word = 0;
|
|
std::memcpy(&word, data + i, sizeof(word));
|
|
hash = (hash ^ word) * 1099511628211ull;
|
|
}
|
|
for (; i < size; ++i) hash = (hash ^ data[i]) * 1099511628211ull;
|
|
hash = (hash ^ size) * 1099511628211ull;
|
|
return hash;
|
|
}
|
|
|
|
// Immediate-mode draws have no source-buffer key. For them, verify content before reusing a slot.
|
|
std::uint64_t geometry_hash(const Render3DDraw& draw) {
|
|
std::uint64_t hash = 14695981039346656037ull;
|
|
const std::uint32_t flags = (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u);
|
|
hash = hash_bytes(hash, &flags, sizeof(flags));
|
|
hash = hash_bytes(hash, draw.positions.data(), draw.positions.size() * sizeof(float));
|
|
hash = hash_bytes(hash, draw.rhw.data(), draw.rhw.size() * sizeof(float));
|
|
hash = hash_bytes(hash, draw.vertex_fog.data(), draw.vertex_fog.size() * sizeof(float));
|
|
hash = hash_bytes(hash, draw.normals.data(), draw.normals.size() * sizeof(float));
|
|
hash = hash_bytes(hash, draw.uv0.data(), draw.uv0.size() * sizeof(float));
|
|
hash = hash_bytes(hash, draw.uv1.data(), draw.uv1.size() * sizeof(float));
|
|
hash = hash_bytes(hash, draw.diffuse.data(), draw.diffuse.size() * sizeof(std::uint32_t));
|
|
return hash_bytes(hash, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t));
|
|
}
|
|
|
|
mtimage::Image decode_tga(const std::uint8_t* data, std::size_t size) {
|
|
mtimage::Image out;
|
|
if (size < 18) return out;
|
|
const std::uint8_t id_len = data[0];
|
|
const std::uint8_t cmap_type = data[1];
|
|
const std::uint8_t img_type = data[2];
|
|
const std::uint32_t w = std::uint32_t(data[12]) | (std::uint32_t(data[13]) << 8);
|
|
const std::uint32_t h = std::uint32_t(data[14]) | (std::uint32_t(data[15]) << 8);
|
|
const std::uint8_t bpp = data[16];
|
|
const std::uint8_t desc = data[17];
|
|
if (cmap_type != 0 || !w || !h || w > 4096 || h > 4096) return out;
|
|
if (img_type != 2 && img_type != 3 && img_type != 10) return out;
|
|
const std::size_t bytes_per_pixel = bpp / 8;
|
|
if (bytes_per_pixel != 1 && bytes_per_pixel != 3 && bytes_per_pixel != 4) return out;
|
|
std::size_t offset = 18 + std::size_t(id_len);
|
|
if (offset > size) return out;
|
|
|
|
const std::size_t pixel_count = std::size_t(w) * std::size_t(h);
|
|
std::vector<std::uint8_t> temp(pixel_count * 4);
|
|
auto write_pixel = [&](std::size_t idx, const std::uint8_t* src) {
|
|
std::uint8_t* dst = &temp[idx * 4];
|
|
if (bytes_per_pixel == 1) {
|
|
dst[0] = dst[1] = dst[2] = src[0];
|
|
dst[3] = 255;
|
|
} else if (bytes_per_pixel == 3) {
|
|
dst[0] = src[2];
|
|
dst[1] = src[1];
|
|
dst[2] = src[0];
|
|
dst[3] = 255;
|
|
} else {
|
|
dst[0] = src[2];
|
|
dst[1] = src[1];
|
|
dst[2] = src[0];
|
|
dst[3] = src[3];
|
|
}
|
|
};
|
|
|
|
if (img_type == 2 || img_type == 3) {
|
|
if (offset + pixel_count * bytes_per_pixel > size) return out;
|
|
for (std::size_t i = 0; i < pixel_count; ++i)
|
|
write_pixel(i, data + offset + i * bytes_per_pixel);
|
|
} else if (img_type == 10) {
|
|
std::size_t i = 0;
|
|
while (i < pixel_count && offset < size) {
|
|
const std::uint8_t header = data[offset++];
|
|
const std::size_t run = (header & 0x7fu) + 1u;
|
|
if (header & 0x80u) {
|
|
if (offset + bytes_per_pixel > size) return out;
|
|
for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i)
|
|
write_pixel(i, data + offset);
|
|
offset += bytes_per_pixel;
|
|
} else {
|
|
if (offset + run * bytes_per_pixel > size) return out;
|
|
for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) {
|
|
write_pixel(i, data + offset);
|
|
offset += bytes_per_pixel;
|
|
}
|
|
}
|
|
}
|
|
if (i != pixel_count) return out;
|
|
}
|
|
|
|
out.w = static_cast<std::uint16_t>(w);
|
|
out.h = static_cast<std::uint16_t>(h);
|
|
const bool top_origin = (desc & 0x20u) != 0;
|
|
if (top_origin) {
|
|
out.rgba = std::move(temp);
|
|
} else {
|
|
out.rgba.resize(pixel_count * 4);
|
|
const std::size_t row_bytes = std::size_t(w) * 4;
|
|
for (std::uint32_t y = 0; y < h; ++y)
|
|
std::memcpy(&out.rgba[std::size_t(y) * row_bytes], &temp[std::size_t(h - 1 - y) * row_bytes], row_bytes);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
mtimage::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size) {
|
|
if (!data || size < 12) return {};
|
|
if (data[0] == 'M' && data[1] == 'T' && data[2] == 'R' && data[3] == 'A') {
|
|
std::uint32_t w = 0, h = 0;
|
|
std::memcpy(&w, data + 4, 4);
|
|
std::memcpy(&h, data + 8, 4);
|
|
const std::size_t bytes = std::size_t(w) * std::size_t(h) * 4;
|
|
if (w > 0 && h > 0 && w <= 4096 && h <= 4096 && size == 12 + bytes) {
|
|
mtimage::Image out;
|
|
out.w = static_cast<std::uint16_t>(w);
|
|
out.h = static_cast<std::uint16_t>(h);
|
|
out.rgba.assign(data + 12, data + 12 + bytes);
|
|
return out;
|
|
}
|
|
return {};
|
|
}
|
|
if (data[0] == 'D' && data[1] == 'D' && data[2] == 'S' && data[3] == ' ')
|
|
return mtimage::load_dds(data, size);
|
|
auto tga = decode_tga(data, size);
|
|
if (tga.ok()) return tga;
|
|
if (size > static_cast<std::size_t>(INT_MAX)) return {};
|
|
int w = 0, h = 0, channels = 0;
|
|
stbi_uc* pixels = stbi_load_from_memory(data, static_cast<int>(size), &w, &h, &channels, 4);
|
|
if (pixels && w > 0 && h > 0 && w <= 4096 && h <= 4096) {
|
|
mtimage::Image out;
|
|
out.w = static_cast<std::uint16_t>(w);
|
|
out.h = static_cast<std::uint16_t>(h);
|
|
out.rgba.assign(pixels, pixels + std::size_t(w) * std::size_t(h) * 4);
|
|
stbi_image_free(pixels);
|
|
return out;
|
|
}
|
|
stbi_image_free(pixels);
|
|
return {};
|
|
}
|
|
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
bool read_live_pack_texture(const std::string& vpath, std::vector<std::uint8_t>& bytes) {
|
|
if (vpath.empty() || !mtpack40250::ready())
|
|
return false;
|
|
std::string norm = vpath;
|
|
for (char& ch : norm)
|
|
if (ch == '\\') ch = '/';
|
|
std::string stripped = norm;
|
|
if (stripped.size() >= 2 && stripped[1] == ':')
|
|
stripped = stripped.substr(2);
|
|
while (!stripped.empty() && stripped.front() == '/')
|
|
stripped.erase(stripped.begin());
|
|
auto lower = [](std::string s) {
|
|
for (char& ch : s)
|
|
if (ch >= 'A' && ch <= 'Z')
|
|
ch = static_cast<char>(ch - 'A' + 'a');
|
|
return s;
|
|
};
|
|
for (const std::string& candidate : {
|
|
norm, stripped, "d:/" + stripped,
|
|
lower(norm), lower(stripped), lower("d:/" + stripped)}) {
|
|
if (mtpack40250::read(candidate, bytes) && !bytes.empty())
|
|
return true;
|
|
}
|
|
return false;
|
|
}
|
|
|
|
SDL_Cursor* load_game_cursor(int shape) {
|
|
static constexpr std::array<const char*, 15> images = {
|
|
"cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub",
|
|
"cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub",
|
|
"cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub",
|
|
"cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"};
|
|
if (shape < 0 || shape >= static_cast<int>(images.size())) return nullptr;
|
|
const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape];
|
|
std::vector<std::uint8_t> sub_bytes;
|
|
if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr;
|
|
std::unordered_map<std::string, std::string> tokens;
|
|
std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end()));
|
|
std::string line;
|
|
while (std::getline(input, line)) {
|
|
std::istringstream fields(line);
|
|
std::string key, value;
|
|
if (!(fields >> key >> value)) continue;
|
|
if (!value.empty() && value.front() == '"') {
|
|
value.erase(0, 1);
|
|
if (!value.empty() && value.back() == '"') value.pop_back();
|
|
}
|
|
std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); });
|
|
tokens[key] = value;
|
|
}
|
|
if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr;
|
|
const std::string image_path = tokens["version"] == "2.0"
|
|
? std::string("d:/ymir work/ui/cursor/") + tokens["image"]
|
|
: std::string("d:/ymir work/ui/") + tokens["image"];
|
|
std::vector<std::uint8_t> image_bytes;
|
|
if (!read_live_pack_texture(image_path, image_bytes)) return nullptr;
|
|
const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size());
|
|
if (!image.ok()) return nullptr;
|
|
const int left = std::atoi(tokens["left"].c_str());
|
|
const int top = std::atoi(tokens["top"].c_str());
|
|
const int right = std::atoi(tokens["right"].c_str());
|
|
const int bottom = std::atoi(tokens["bottom"].c_str());
|
|
if (left < 0 || top < 0 || right <= left || bottom <= top ||
|
|
right > image.w || bottom > image.h) return nullptr;
|
|
const int width = right - left, height = bottom - top;
|
|
std::vector<std::uint8_t> pixels(std::size_t(width) * height * 4);
|
|
for (int y = 0; y < height; ++y)
|
|
std::memcpy(pixels.data() + std::size_t(y) * width * 4,
|
|
image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4,
|
|
std::size_t(width) * 4);
|
|
SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32,
|
|
pixels.data(), width * 4);
|
|
if (!surface) return nullptr;
|
|
const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0;
|
|
const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0;
|
|
SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y);
|
|
SDL_DestroySurface(surface);
|
|
return cursor;
|
|
}
|
|
|
|
class NativeAudioEngine {
|
|
struct MemoryAudioBuffer {
|
|
const std::uint8_t* data = nullptr;
|
|
std::size_t size = 0;
|
|
};
|
|
struct Voice {
|
|
const std::vector<float>* samples = nullptr;
|
|
std::size_t frame_cursor = 0;
|
|
float volume = 1.0f;
|
|
float target_volume = 1.0f;
|
|
float fade_step_per_frame = 0.0f;
|
|
bool stop_after_fade = false;
|
|
bool loop = false;
|
|
bool is_3d = false;
|
|
int handle = 0;
|
|
std::string filename;
|
|
};
|
|
|
|
#ifdef __APPLE__
|
|
static OSStatus mem_audio_read_proc(
|
|
void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) {
|
|
const auto* mem = static_cast<const MemoryAudioBuffer*>(inClientData);
|
|
if (inPosition < 0 || static_cast<std::size_t>(inPosition) >= mem->size) {
|
|
*actualCount = 0;
|
|
return noErr;
|
|
}
|
|
const std::size_t avail = mem->size - static_cast<std::size_t>(inPosition);
|
|
const std::size_t to_read = std::min<std::size_t>(requestCount, avail);
|
|
std::memcpy(buffer, mem->data + inPosition, to_read);
|
|
*actualCount = static_cast<UInt32>(to_read);
|
|
return noErr;
|
|
}
|
|
|
|
static SInt64 mem_audio_get_size_proc(void* inClientData) {
|
|
return static_cast<SInt64>(static_cast<const MemoryAudioBuffer*>(inClientData)->size);
|
|
}
|
|
#endif
|
|
|
|
public:
|
|
NativeAudioEngine() {
|
|
if (SDL_InitSubSystem(SDL_INIT_AUDIO)) {
|
|
SDL_AudioSpec spec{};
|
|
spec.format = SDL_AUDIO_F32;
|
|
spec.channels = 2;
|
|
spec.freq = 44100;
|
|
stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_PLAYBACK, &spec, nullptr, nullptr);
|
|
if (stream_)
|
|
SDL_ResumeAudioStreamDevice(stream_);
|
|
}
|
|
// Always DrainAudioCommands() from cold boot so stale queued commands don't accumulate.
|
|
(void)DrainAudioCommands();
|
|
}
|
|
|
|
~NativeAudioEngine() {
|
|
if (stream_)
|
|
SDL_DestroyAudioStream(stream_);
|
|
SDL_QuitSubSystem(SDL_INIT_AUDIO);
|
|
}
|
|
|
|
void pump() {
|
|
const auto commands = DrainAudioCommands();
|
|
for (const auto& cmd : commands) {
|
|
++commands_drained_;
|
|
switch (cmd.type) {
|
|
case AudioCommand::PlaySound2D: {
|
|
const auto* clip = get_clip(cmd.filename);
|
|
if (clip && !clip->empty()) {
|
|
Voice v{};
|
|
v.samples = clip;
|
|
v.volume = sound_volume_;
|
|
v.target_volume = sound_volume_;
|
|
v.filename = cmd.filename;
|
|
voices_.push_back(std::move(v));
|
|
++played_2d_;
|
|
}
|
|
break;
|
|
}
|
|
case AudioCommand::PlaySound3D: {
|
|
const auto* clip = get_clip(cmd.filename);
|
|
if (clip && !clip->empty()) {
|
|
Voice v{};
|
|
v.samples = clip;
|
|
v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
|
|
v.target_volume = v.volume;
|
|
v.loop = cmd.play_count != 1;
|
|
v.is_3d = true;
|
|
v.handle = cmd.id;
|
|
v.filename = cmd.filename;
|
|
voices_.push_back(std::move(v));
|
|
++played_3d_;
|
|
}
|
|
break;
|
|
}
|
|
case AudioCommand::StopSound3D:
|
|
voices_.erase(
|
|
std::remove_if(voices_.begin(), voices_.end(), [&](const Voice& v) {
|
|
return v.is_3d && v.handle == cmd.id;
|
|
}),
|
|
voices_.end());
|
|
break;
|
|
case AudioCommand::StopAllSound3D:
|
|
voices_.erase(
|
|
std::remove_if(voices_.begin(), voices_.end(), [](const Voice& v) { return v.is_3d; }),
|
|
voices_.end());
|
|
break;
|
|
case AudioCommand::SetSoundVolume3D:
|
|
for (auto& v : voices_) {
|
|
if (v.is_3d && v.handle == cmd.id) {
|
|
v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
|
|
v.target_volume = v.volume;
|
|
}
|
|
}
|
|
break;
|
|
case AudioCommand::PlayMusic:
|
|
case AudioCommand::FadeInMusic: {
|
|
const auto* clip = get_clip(cmd.filename);
|
|
if (clip && !clip->empty()) {
|
|
bgm_.samples = clip;
|
|
bgm_.frame_cursor = 0;
|
|
bgm_.loop = true;
|
|
bgm_.filename = cmd.filename;
|
|
bgm_.stop_after_fade = false;
|
|
const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
|
|
bgm_.target_volume = target;
|
|
if (cmd.type == AudioCommand::FadeInMusic) {
|
|
bgm_.volume = 0.0f;
|
|
bgm_.fade_step_per_frame = target / (44100.0f * 1.5f);
|
|
} else {
|
|
bgm_.volume = target;
|
|
bgm_.fade_step_per_frame = 0.0f;
|
|
}
|
|
++played_music_;
|
|
}
|
|
break;
|
|
}
|
|
case AudioCommand::FadeOutMusic:
|
|
case AudioCommand::FadeOutAllMusic:
|
|
if (bgm_.samples) {
|
|
bgm_.target_volume = 0.0f;
|
|
bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f);
|
|
bgm_.stop_after_fade = true;
|
|
}
|
|
break;
|
|
case AudioCommand::FadeLimitOutMusic:
|
|
if (bgm_.samples) {
|
|
bgm_.target_volume = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
|
|
bgm_.fade_step_per_frame = (bgm_.target_volume - bgm_.volume) / (44100.0f * 1.0f);
|
|
bgm_.stop_after_fade = false;
|
|
}
|
|
break;
|
|
case AudioCommand::SetMusicVolume:
|
|
music_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f);
|
|
if (bgm_.samples && !bgm_.stop_after_fade) {
|
|
bgm_.volume = music_volume_;
|
|
bgm_.target_volume = music_volume_;
|
|
}
|
|
break;
|
|
case AudioCommand::SetSoundVolume:
|
|
sound_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f);
|
|
break;
|
|
default:
|
|
break;
|
|
}
|
|
}
|
|
|
|
if (!stream_) return;
|
|
const int queued_bytes = SDL_GetAudioStreamAvailable(stream_);
|
|
if (queued_bytes < 0) return;
|
|
const std::size_t queued_frames = static_cast<std::size_t>(queued_bytes) / (2 * sizeof(float));
|
|
constexpr std::size_t kTargetQueuedFrames = 4410; // ~100 ms stereo @ 44.1 kHz
|
|
if (queued_frames >= kTargetQueuedFrames) return;
|
|
const std::size_t frames_to_mix = kTargetQueuedFrames - queued_frames;
|
|
mix_buffer_.assign(frames_to_mix * 2, 0.0f);
|
|
|
|
auto mix_voice = [&](Voice& v) -> bool {
|
|
if (!v.samples || v.samples->empty()) return false;
|
|
const std::size_t total_frames = v.samples->size() / 2;
|
|
if (total_frames == 0) return false;
|
|
const float* src = v.samples->data();
|
|
for (std::size_t f = 0; f < frames_to_mix; ++f) {
|
|
if (v.frame_cursor >= total_frames) {
|
|
if (v.loop) v.frame_cursor = 0;
|
|
else return false;
|
|
}
|
|
if (v.fade_step_per_frame != 0.0f) {
|
|
v.volume += v.fade_step_per_frame;
|
|
if ((v.fade_step_per_frame > 0.0f && v.volume >= v.target_volume) ||
|
|
(v.fade_step_per_frame < 0.0f && v.volume <= v.target_volume)) {
|
|
v.volume = v.target_volume;
|
|
v.fade_step_per_frame = 0.0f;
|
|
if (v.stop_after_fade && v.volume <= 0.0001f) {
|
|
v.samples = nullptr;
|
|
return false;
|
|
}
|
|
}
|
|
}
|
|
mix_buffer_[f * 2 + 0] += src[v.frame_cursor * 2 + 0] * v.volume;
|
|
mix_buffer_[f * 2 + 1] += src[v.frame_cursor * 2 + 1] * v.volume;
|
|
++v.frame_cursor;
|
|
}
|
|
return v.loop || v.frame_cursor < total_frames;
|
|
};
|
|
|
|
if (bgm_.samples) {
|
|
if (!mix_voice(bgm_)) bgm_.samples = nullptr;
|
|
}
|
|
for (auto it = voices_.begin(); it != voices_.end();) {
|
|
if (!mix_voice(*it)) it = voices_.erase(it);
|
|
else ++it;
|
|
}
|
|
for (float& sample : mix_buffer_)
|
|
sample = std::clamp(sample, -1.0f, 1.0f);
|
|
SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast<int>(mix_buffer_.size() * sizeof(float)));
|
|
}
|
|
|
|
private:
|
|
const std::vector<float>* get_clip(const std::string& vpath) {
|
|
if (vpath.empty()) return nullptr;
|
|
auto [it, inserted] = clips_.try_emplace(vpath);
|
|
if (!inserted) return &it->second;
|
|
#ifdef __APPLE__
|
|
std::vector<std::uint8_t> bytes;
|
|
if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16)
|
|
return &it->second;
|
|
MemoryAudioBuffer mem{bytes.data(), bytes.size()};
|
|
AudioFileID audio_file = nullptr;
|
|
if (AudioFileOpenWithCallbacks(
|
|
&mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr ||
|
|
!audio_file) {
|
|
return &it->second;
|
|
}
|
|
ExtAudioFileRef ext_file = nullptr;
|
|
if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) {
|
|
AudioFileClose(audio_file);
|
|
return &it->second;
|
|
}
|
|
AudioStreamBasicDescription client_format{};
|
|
client_format.mSampleRate = 44100.0;
|
|
client_format.mFormatID = kAudioFormatLinearPCM;
|
|
client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked;
|
|
client_format.mBytesPerPacket = 8;
|
|
client_format.mFramesPerPacket = 1;
|
|
client_format.mBytesPerFrame = 8;
|
|
client_format.mChannelsPerFrame = 2;
|
|
client_format.mBitsPerChannel = 32;
|
|
if (ExtAudioFileSetProperty(
|
|
ext_file,
|
|
kExtAudioFileProperty_ClientDataFormat,
|
|
sizeof(client_format),
|
|
&client_format) == noErr) {
|
|
std::vector<float> chunk(4096 * 2);
|
|
while (true) {
|
|
UInt32 frame_count = 4096;
|
|
AudioBufferList buf_list{};
|
|
buf_list.mNumberBuffers = 1;
|
|
buf_list.mBuffers[0].mNumberChannels = 2;
|
|
buf_list.mBuffers[0].mDataByteSize = static_cast<UInt32>(chunk.size() * sizeof(float));
|
|
buf_list.mBuffers[0].mData = chunk.data();
|
|
if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0)
|
|
break;
|
|
it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2);
|
|
if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes
|
|
break;
|
|
}
|
|
}
|
|
ExtAudioFileDispose(ext_file);
|
|
AudioFileClose(audio_file);
|
|
#endif
|
|
return &it->second;
|
|
}
|
|
|
|
SDL_AudioStream* stream_ = nullptr;
|
|
std::unordered_map<std::string, std::vector<float>> clips_;
|
|
std::vector<Voice> voices_;
|
|
Voice bgm_{};
|
|
std::vector<float> mix_buffer_;
|
|
float sound_volume_ = 0.8f;
|
|
float music_volume_ = 0.5f;
|
|
std::size_t commands_drained_ = 0;
|
|
std::size_t played_2d_ = 0;
|
|
std::size_t played_3d_ = 0;
|
|
std::size_t played_music_ = 0;
|
|
};
|
|
#endif
|
|
|
|
std::string bundled_file_path(const char* name) {
|
|
const char* base = SDL_GetBasePath();
|
|
return std::string(base ? base : "./") + name;
|
|
}
|
|
|
|
#ifdef __ANDROID__
|
|
// MainActivity.performHaptic(): the long-press vibration of the touch gestures.
|
|
void android_haptic() {
|
|
auto* env = static_cast<JNIEnv*>(SDL_GetAndroidJNIEnv());
|
|
auto activity = static_cast<jobject>(SDL_GetAndroidActivity());
|
|
if (!env || !activity) return;
|
|
jclass cls = env->GetObjectClass(activity);
|
|
if (jmethodID method = env->GetMethodID(cls, "performHaptic", "()V"))
|
|
env->CallVoidMethod(activity, method);
|
|
if (env->ExceptionCheck()) env->ExceptionClear();
|
|
env->DeleteLocalRef(cls);
|
|
env->DeleteLocalRef(activity);
|
|
}
|
|
#endif
|
|
|
|
std::vector<std::uint32_t> read_spirv(const char* name) {
|
|
std::size_t size = 0;
|
|
void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size);
|
|
#ifdef __ANDROID__
|
|
if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size);
|
|
if (!bytes) bytes = SDL_LoadFile(name, &size);
|
|
#endif
|
|
if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name);
|
|
if (size == 0 || size % 4) {
|
|
SDL_free(bytes);
|
|
throw std::runtime_error(std::string("invalid SPIR-V size: ") + name);
|
|
}
|
|
std::vector<std::uint32_t> words(size / 4);
|
|
std::memcpy(words.data(), bytes, size);
|
|
SDL_free(bytes);
|
|
return words;
|
|
}
|
|
|
|
class VulkanWindow {
|
|
struct FrameSlot;
|
|
struct GeometryId {
|
|
std::uint64_t key = 0, signature = 0;
|
|
bool operator==(const GeometryId&) const = default;
|
|
};
|
|
struct GeometryIdHash {
|
|
std::size_t operator()(GeometryId id) const {
|
|
return std::size_t(id.key ^ (id.signature + 0x9e3779b97f4a7c15ull + (id.key << 6) + (id.key >> 2)));
|
|
}
|
|
};
|
|
struct Geometry {
|
|
VkBuffer buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory memory = VK_NULL_HANDLE;
|
|
VkDeviceSize index_offset = 0;
|
|
std::uint64_t last_used_frame = 0;
|
|
std::uint32_t vertex_count = 0, index_count = 0;
|
|
};
|
|
struct GpuTexture {
|
|
VkImage image = VK_NULL_HANDLE;
|
|
VkDeviceMemory memory = VK_NULL_HANDLE;
|
|
VkImageView view = VK_NULL_HANDLE;
|
|
VkDescriptorSet descriptor = VK_NULL_HANDLE;
|
|
VkDescriptorSet ui_descriptor = VK_NULL_HANDLE;
|
|
std::uint64_t last_used_frame = 0;
|
|
std::uint64_t last_decode_attempt_frame = 0;
|
|
};
|
|
struct PairedDescriptor {
|
|
VkDescriptorSet set = VK_NULL_HANDLE;
|
|
std::uint64_t last_used_frame = 0;
|
|
};
|
|
// An offscreen D3D render-target texture (the 40250 character shadow map, "rt:<id>:<w>x<h>"):
|
|
// drawn by the frame's offscreen pass ahead of the back-buffer pass and sampled by that pass.
|
|
// Between frames the colour image stays SHADER_READ_ONLY_OPTIMAL and the depth image
|
|
// DEPTH_STENCIL_ATTACHMENT_OPTIMAL; the render pass keeps (LOAD/STORE) both, like a D3D surface.
|
|
struct RenderTarget {
|
|
GpuTexture texture; // colour image, view and descriptors used for sampling
|
|
VkImage depth_image = VK_NULL_HANDLE;
|
|
VkDeviceMemory depth_memory = VK_NULL_HANDLE;
|
|
VkImageView depth_view = VK_NULL_HANDLE;
|
|
VkFramebuffer framebuffer = VK_NULL_HANDLE;
|
|
std::uint32_t width = 0, height = 0;
|
|
bool initialized = false; // cleared to white / depth 1 by the first frame that uses it
|
|
};
|
|
struct PreparedDraw {
|
|
Geometry* geometry = nullptr;
|
|
VkPipeline pipeline = VK_NULL_HANDLE;
|
|
VkDescriptorSet descriptor = VK_NULL_HANDLE;
|
|
PushConstants constants{};
|
|
FixedFunctionState fixed_state{};
|
|
std::uint32_t state_offset = 0;
|
|
VkViewport viewport{};
|
|
// Offscreen render target of the draw, or null for the back buffer.
|
|
RenderTarget* target = nullptr;
|
|
// IDirect3DDevice8::Clear recorded in draw order; geometry is null for these.
|
|
std::uint32_t clear_flags = 0;
|
|
VkClearValue clear_color{};
|
|
VkClearValue clear_depth{};
|
|
VkRect2D clear_rect{};
|
|
};
|
|
struct UiBatch {
|
|
std::uint32_t first_index = 0;
|
|
std::uint32_t index_count = 0;
|
|
VkPipeline pipeline = VK_NULL_HANDLE;
|
|
VkDescriptorSet descriptor = VK_NULL_HANDLE;
|
|
PushConstants constants{};
|
|
bool behind_3d = false;
|
|
std::uint32_t state_offset = 0;
|
|
};
|
|
|
|
static constexpr std::size_t kMaxBonesPerFrame = 65536;
|
|
static constexpr VkDeviceSize kBoneBufferBytes = kMaxBonesPerFrame * 16 * sizeof(float);
|
|
static constexpr std::size_t kMaxUiVerticesPerFrame = 65536;
|
|
static constexpr std::size_t kMaxUiIndicesPerFrame = 98304;
|
|
static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex);
|
|
static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t);
|
|
static constexpr VkDeviceSize kStagingRingBytes = 4ull * 1024 * 1024;
|
|
static constexpr VkDeviceSize kStagingRingMaxBytes = 64ull * 1024 * 1024;
|
|
|
|
public:
|
|
struct Timings {
|
|
double sync_ms = 0;
|
|
double fence_ms = 0; // inside sync_ms: waiting for the frame slot's previous GPU work
|
|
double prepare_ms = 0;
|
|
double steady_prepare_ms = 0;
|
|
double submit_ms = 0;
|
|
double present_ms = 0;
|
|
double gpu_ms = 0;
|
|
std::uint64_t gpu_samples = 0;
|
|
// Inside prepare_ms: creating geometry buffers and the blocking texture uploads.
|
|
double geometry_upload_ms = 0;
|
|
double texture_upload_ms = 0;
|
|
};
|
|
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
void enable_game_hardware_cursor() {
|
|
hardware_cursor_enabled_ = true;
|
|
for (int shape = 0; shape < static_cast<int>(game_cursors_.size()); ++shape) {
|
|
game_cursors_[shape] = load_game_cursor(shape);
|
|
}
|
|
fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT);
|
|
SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_);
|
|
SDL_ShowCursor();
|
|
os_cursor_hidden_ = false;
|
|
}
|
|
|
|
void sync_game_cursor() {
|
|
if (!hardware_cursor_enabled_) return;
|
|
const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() &&
|
|
!PythonBoot::IsSoftwareCursorVisible();
|
|
if (visible == os_cursor_hidden_) {
|
|
if (visible) SDL_ShowCursor();
|
|
else SDL_HideCursor();
|
|
os_cursor_hidden_ = !visible;
|
|
}
|
|
if (!visible) return;
|
|
const int shape = PythonBoot::CursorShape();
|
|
if (shape == current_cursor_shape_) return;
|
|
SDL_Cursor* cursor = shape >= 0 && shape < static_cast<int>(game_cursors_.size())
|
|
? game_cursors_[shape] : nullptr;
|
|
if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_);
|
|
current_cursor_shape_ = shape;
|
|
}
|
|
#endif
|
|
|
|
explicit VulkanWindow(bool vsync = true, int init_width = 960, int init_height = 640) : vsync_(vsync) {
|
|
#ifdef __APPLE__
|
|
if (!std::getenv("VK_ICD_FILENAMES")) {
|
|
for (const char* icd : {
|
|
"/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json",
|
|
"/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json"}) {
|
|
if (std::ifstream(icd).good()) {
|
|
setenv("VK_ICD_FILENAMES", icd, 0);
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
#endif
|
|
// Android's back key reaches the game as SDL_SCANCODE_AC_BACK (-> ESC) instead of finishing the activity.
|
|
SDL_SetHint(SDL_HINT_ANDROID_TRAP_BACK_BUTTON, "1");
|
|
if (!SDL_Init(SDL_INIT_VIDEO)) throw std::runtime_error(SDL_GetError());
|
|
SDL_SetHint(SDL_HINT_ORIENTATIONS, "LandscapeLeft LandscapeRight");
|
|
window_ = SDL_CreateWindow(
|
|
"Metin2 Native Vulkan Client",
|
|
std::max(init_width, 320),
|
|
std::max(init_height, 240),
|
|
SDL_WINDOW_VULKAN | SDL_WINDOW_RESIZABLE | SDL_WINDOW_HIGH_PIXEL_DENSITY);
|
|
if (!window_) throw std::runtime_error(SDL_GetError());
|
|
SDL_StartTextInput(window_);
|
|
Uint32 extension_count = 0;
|
|
const char* const* sdl_extensions = SDL_Vulkan_GetInstanceExtensions(&extension_count);
|
|
if (!sdl_extensions) throw std::runtime_error(SDL_GetError());
|
|
std::vector<const char*> extensions(sdl_extensions, sdl_extensions + extension_count);
|
|
std::uint32_t available_count = 0;
|
|
check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, nullptr), "vkEnumerateInstanceExtensionProperties count");
|
|
std::vector<VkExtensionProperties> available(available_count);
|
|
check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, available.data()), "vkEnumerateInstanceExtensionProperties");
|
|
const bool portability = std::any_of(available.begin(), available.end(), [](const auto& extension) {
|
|
return std::strcmp(extension.extensionName, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0;
|
|
});
|
|
if (portability && std::find_if(extensions.begin(), extensions.end(), [](const char* name) {
|
|
return std::strcmp(name, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0;
|
|
}) == extensions.end())
|
|
extensions.push_back(VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME);
|
|
VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO};
|
|
app.pApplicationName = "Metin2 native renderer";
|
|
app.apiVersion = VK_API_VERSION_1_1;
|
|
VkInstanceCreateInfo instance_info{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO};
|
|
instance_info.flags = portability ? VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR : 0;
|
|
instance_info.pApplicationInfo = &app;
|
|
instance_info.enabledExtensionCount = static_cast<std::uint32_t>(extensions.size());
|
|
instance_info.ppEnabledExtensionNames = extensions.data();
|
|
check(vkCreateInstance(&instance_info, nullptr, &instance_), "vkCreateInstance");
|
|
if (!SDL_Vulkan_CreateSurface(window_, instance_, nullptr, &surface_)) throw std::runtime_error(SDL_GetError());
|
|
select_device();
|
|
VkCommandPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO};
|
|
pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT;
|
|
pool_info.queueFamilyIndex = queue_family_;
|
|
check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool");
|
|
VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO};
|
|
query_info.queryType = VK_QUERY_TYPE_TIMESTAMP;
|
|
query_info.queryCount = 2 * kMaxFramesInFlight;
|
|
check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool");
|
|
VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
|
|
VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO};
|
|
fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT;
|
|
for (std::uint32_t i = 0; i < kMaxFramesInFlight; ++i) {
|
|
auto& slot = frames_[i];
|
|
VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
|
|
alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1;
|
|
check(vkAllocateCommandBuffers(device_, &alloc, &slot.command), "vkAllocateCommandBuffers");
|
|
check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &slot.acquire), "vkCreateSemaphore acquire");
|
|
check(vkCreateFence(device_, &fence_info, nullptr, &slot.fence), "vkCreateFence");
|
|
slot.query_base = 2 * i;
|
|
}
|
|
create_swapchain();
|
|
create_descriptors_and_buffers();
|
|
create_render_pass_and_layout();
|
|
touch_controller.update_screen_size(int(width()), int(height()));
|
|
}
|
|
|
|
~VulkanWindow() {
|
|
if (hardware_cursor_enabled_) {
|
|
SDL_SetCursor(nullptr);
|
|
for (SDL_Cursor* cursor : game_cursors_) if (cursor) SDL_DestroyCursor(cursor);
|
|
if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_);
|
|
}
|
|
if (device_) vkDeviceWaitIdle(device_);
|
|
for (auto& slot : frames_) {
|
|
release_retired_buffers(slot);
|
|
if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory);
|
|
if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr);
|
|
if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr);
|
|
if (slot.bone_mapped) vkUnmapMemory(device_, slot.bone_memory);
|
|
if (slot.bone_buffer) vkDestroyBuffer(device_, slot.bone_buffer, nullptr);
|
|
if (slot.bone_memory) vkFreeMemory(device_, slot.bone_memory, nullptr);
|
|
if (slot.state_mapped) vkUnmapMemory(device_, slot.state_memory);
|
|
if (slot.state_buffer) vkDestroyBuffer(device_, slot.state_buffer, nullptr);
|
|
if (slot.state_memory) vkFreeMemory(device_, slot.state_memory, nullptr);
|
|
if (slot.ui_mapped) vkUnmapMemory(device_, slot.ui_memory);
|
|
if (slot.ui_buffer) vkDestroyBuffer(device_, slot.ui_buffer, nullptr);
|
|
if (slot.ui_memory) vkFreeMemory(device_, slot.ui_memory, nullptr);
|
|
if (slot.fence) vkDestroyFence(device_, slot.fence, nullptr);
|
|
if (slot.acquire) vkDestroySemaphore(device_, slot.acquire, nullptr);
|
|
}
|
|
if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr);
|
|
for (VkSemaphore semaphore : rendered_) vkDestroySemaphore(device_, semaphore, nullptr);
|
|
for (auto& entry : geometries_) release_geometry(entry.second);
|
|
for (auto& entry : textures_) release_texture(entry.second);
|
|
for (auto& entry : render_targets_) release_render_target(entry.second);
|
|
if (offscreen_pass_) vkDestroyRenderPass(device_, offscreen_pass_, nullptr);
|
|
release_texture(fallback_texture_);
|
|
for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr);
|
|
if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr);
|
|
if (fragment_module_) vkDestroyShaderModule(device_, fragment_module_, nullptr);
|
|
if (layout_) vkDestroyPipelineLayout(device_, layout_, nullptr);
|
|
if (descriptor_pool_) vkDestroyDescriptorPool(device_, descriptor_pool_, nullptr);
|
|
if (descriptor_layout_) vkDestroyDescriptorSetLayout(device_, descriptor_layout_, nullptr);
|
|
if (bone_descriptor_layout_) vkDestroyDescriptorSetLayout(device_, bone_descriptor_layout_, nullptr);
|
|
if (sampler_) vkDestroySampler(device_, sampler_, nullptr);
|
|
if (clamp_sampler_) vkDestroySampler(device_, clamp_sampler_, nullptr);
|
|
if (ui_sampler_) vkDestroySampler(device_, ui_sampler_, nullptr);
|
|
for (const auto& entry : state_samplers_) vkDestroySampler(device_, entry.second, nullptr);
|
|
for (auto framebuffer : framebuffers_) vkDestroyFramebuffer(device_, framebuffer, nullptr);
|
|
if (pass_) vkDestroyRenderPass(device_, pass_, nullptr);
|
|
if (depth_view_) vkDestroyImageView(device_, depth_view_, nullptr);
|
|
if (depth_image_) vkDestroyImage(device_, depth_image_, nullptr);
|
|
if (depth_memory_) vkFreeMemory(device_, depth_memory_, nullptr);
|
|
destroy_msaa_color();
|
|
for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr);
|
|
if (swapchain_) vkDestroySwapchainKHR(device_, swapchain_, nullptr);
|
|
if (pool_) vkDestroyCommandPool(device_, pool_, nullptr);
|
|
if (device_) vkDestroyDevice(device_, nullptr);
|
|
if (surface_) vkDestroySurfaceKHR(instance_, surface_, nullptr);
|
|
if (instance_) vkDestroyInstance(instance_, nullptr);
|
|
if (window_) {
|
|
SDL_StopTextInput(window_);
|
|
SDL_DestroyWindow(window_);
|
|
}
|
|
SDL_Quit();
|
|
}
|
|
|
|
// 32-bit bottom-up BMP from a B8G8R8A8 / R8G8B8A8 swapchain readback.
|
|
void write_bmp(const std::string& path, const std::uint8_t* pixels) const {
|
|
const std::uint32_t w = extent_.width, h = extent_.height;
|
|
const bool rgba = swapchain_format_ == VK_FORMAT_R8G8B8A8_UNORM || swapchain_format_ == VK_FORMAT_R8G8B8A8_SRGB;
|
|
std::vector<std::uint8_t> file(54 + std::size_t(w) * h * 4);
|
|
auto put32 = [&](std::size_t at, std::uint32_t value) { std::memcpy(&file[at], &value, 4); };
|
|
file[0] = 'B'; file[1] = 'M';
|
|
put32(2, std::uint32_t(file.size())); put32(10, 54); put32(14, 40);
|
|
put32(18, w); put32(22, h); put32(26, 1u | (32u << 16)); put32(34, w * h * 4);
|
|
for (std::uint32_t y = 0; y < h; ++y)
|
|
for (std::uint32_t x = 0; x < w; ++x) {
|
|
const std::uint8_t* src = pixels + (std::size_t(y) * w + x) * 4;
|
|
std::uint8_t* dst = &file[54 + (std::size_t(h - 1 - y) * w + x) * 4];
|
|
dst[0] = rgba ? src[2] : src[0]; dst[1] = src[1]; dst[2] = rgba ? src[0] : src[2]; dst[3] = 255;
|
|
}
|
|
std::ofstream out(path, std::ios::binary);
|
|
out.write(reinterpret_cast<const char*>(file.data()), std::streamsize(file.size()));
|
|
}
|
|
|
|
void request_screenshot(std::string path, std::uint64_t frame) {
|
|
screenshot_path_ = std::move(path);
|
|
screenshot_frame_ = frame;
|
|
}
|
|
std::uint64_t frame_number() const { return frame_number_; }
|
|
|
|
int msaa_samples() const { return int(samples_); }
|
|
std::uint32_t width() const { return extent_.width; }
|
|
std::uint32_t height() const { return extent_.height; }
|
|
std::uint32_t logical_width() const {
|
|
int w = 0, h = 0;
|
|
if (window_) SDL_GetWindowSize(window_, &w, &h);
|
|
#ifdef __ANDROID__
|
|
// Keep the 40250 UI at the same height as the macOS 1280x800 window,
|
|
// while retaining the device aspect ratio on wider phone displays.
|
|
if (w > 0 && h > 0) return std::max(1, int(std::lround(double(w) * 800.0 / h)));
|
|
#endif
|
|
return w > 0 ? static_cast<std::uint32_t>(w) : extent_.width;
|
|
}
|
|
std::uint32_t logical_height() const {
|
|
int w = 0, h = 0;
|
|
if (window_) SDL_GetWindowSize(window_, &w, &h);
|
|
#ifdef __ANDROID__
|
|
if (w > 0 && h > 0) return 800;
|
|
#endif
|
|
return h > 0 ? static_cast<std::uint32_t>(h) : extent_.height;
|
|
}
|
|
|
|
VkViewport full_viewport() const {
|
|
return {0, 0, float(extent_.width), float(extent_.height), 0, 1};
|
|
}
|
|
|
|
// Window pixels kept clear at the left and right screen edges (rounded corners, cutouts);
|
|
// MainActivity measures them and passes --safe-inset-px.
|
|
void set_safe_inset_px(int px) { safe_inset_px_ = std::max(0, px); }
|
|
// Test builds: frames per second in the top-left corner (--show-fps).
|
|
void set_show_fps(bool show) { show_fps_ = show; }
|
|
SDL_Window* sdl_window() const { return window_; }
|
|
// The last render()'s blocking time: slot fence + swapchain acquire.
|
|
double last_sync_ms() const { return last_sync_ms_; }
|
|
// The same inset in logical UI pixels: the 40250 window layers are shifted and narrowed by it.
|
|
int ui_safe_inset() const {
|
|
int pw = 0, ph = 0;
|
|
if (safe_inset_px_ <= 0 || !window_ || !SDL_GetWindowSizeInPixels(window_, &pw, &ph) || ph <= 0) return 0;
|
|
return int(std::lround(double(safe_inset_px_) * logical_height() / ph));
|
|
}
|
|
|
|
// D3DVIEWPORT8 (in the game's logical screen pixels) scaled to the swapchain extent.
|
|
VkViewport draw_viewport(const Render3DDraw& draw) const {
|
|
if (draw.viewport[2] <= 0 || draw.viewport[3] <= 0) return full_viewport();
|
|
const float sx = float(extent_.width) / float(logical_width());
|
|
const float sy = float(extent_.height) / float(logical_height());
|
|
return {(draw.viewport[0] + draw.ui_offset_x) * sx, draw.viewport[1] * sy, draw.viewport[2] * sx, draw.viewport[3] * sy,
|
|
draw.viewport_z[0], draw.viewport_z[1]};
|
|
}
|
|
|
|
TouchController touch_controller;
|
|
|
|
static int sdl_scancode_to_dik(SDL_Scancode sc) {
|
|
switch (sc) {
|
|
case SDL_SCANCODE_ESCAPE: return 0x01;
|
|
case SDL_SCANCODE_1: return 0x02;
|
|
case SDL_SCANCODE_2: return 0x03;
|
|
case SDL_SCANCODE_3: return 0x04;
|
|
case SDL_SCANCODE_4: return 0x05;
|
|
case SDL_SCANCODE_5: return 0x06;
|
|
case SDL_SCANCODE_6: return 0x07;
|
|
case SDL_SCANCODE_7: return 0x08;
|
|
case SDL_SCANCODE_8: return 0x09;
|
|
case SDL_SCANCODE_9: return 0x0A;
|
|
case SDL_SCANCODE_0: return 0x0B;
|
|
case SDL_SCANCODE_MINUS: return 0x0C;
|
|
case SDL_SCANCODE_EQUALS: return 0x0D;
|
|
case SDL_SCANCODE_BACKSPACE: return 0x0E;
|
|
case SDL_SCANCODE_TAB: return 0x0F;
|
|
case SDL_SCANCODE_Q: return 0x10;
|
|
case SDL_SCANCODE_W: return 0x11;
|
|
case SDL_SCANCODE_E: return 0x12;
|
|
case SDL_SCANCODE_R: return 0x13;
|
|
case SDL_SCANCODE_T: return 0x14;
|
|
case SDL_SCANCODE_Y: return 0x15;
|
|
case SDL_SCANCODE_U: return 0x16;
|
|
case SDL_SCANCODE_I: return 0x17;
|
|
case SDL_SCANCODE_O: return 0x18;
|
|
case SDL_SCANCODE_P: return 0x19;
|
|
case SDL_SCANCODE_LEFTBRACKET: return 0x1A;
|
|
case SDL_SCANCODE_RIGHTBRACKET: return 0x1B;
|
|
case SDL_SCANCODE_RETURN: return 0x1C;
|
|
case SDL_SCANCODE_LCTRL: return 0x1D;
|
|
case SDL_SCANCODE_A: return 0x1E;
|
|
case SDL_SCANCODE_S: return 0x1F;
|
|
case SDL_SCANCODE_D: return 0x20;
|
|
case SDL_SCANCODE_F: return 0x21;
|
|
case SDL_SCANCODE_G: return 0x22;
|
|
case SDL_SCANCODE_H: return 0x23;
|
|
case SDL_SCANCODE_J: return 0x24;
|
|
case SDL_SCANCODE_K: return 0x25;
|
|
case SDL_SCANCODE_L: return 0x26;
|
|
case SDL_SCANCODE_SEMICOLON: return 0x27;
|
|
case SDL_SCANCODE_APOSTROPHE: return 0x28;
|
|
case SDL_SCANCODE_GRAVE: return 0x29;
|
|
case SDL_SCANCODE_LSHIFT: return 0x2A;
|
|
case SDL_SCANCODE_BACKSLASH: return 0x2B;
|
|
case SDL_SCANCODE_Z: return 0x2C;
|
|
case SDL_SCANCODE_X: return 0x2D;
|
|
case SDL_SCANCODE_C: return 0x2E;
|
|
case SDL_SCANCODE_V: return 0x2F;
|
|
case SDL_SCANCODE_B: return 0x30;
|
|
case SDL_SCANCODE_N: return 0x31;
|
|
case SDL_SCANCODE_M: return 0x32;
|
|
case SDL_SCANCODE_COMMA: return 0x33;
|
|
case SDL_SCANCODE_PERIOD: return 0x34;
|
|
case SDL_SCANCODE_SLASH: return 0x35;
|
|
case SDL_SCANCODE_RSHIFT: return 0x36;
|
|
case SDL_SCANCODE_KP_MULTIPLY: return 0x37;
|
|
case SDL_SCANCODE_LALT: return 0x38;
|
|
case SDL_SCANCODE_SPACE: return 0x39;
|
|
case SDL_SCANCODE_CAPSLOCK: return 0x3A;
|
|
case SDL_SCANCODE_F1: return 0x3B;
|
|
case SDL_SCANCODE_F2: return 0x3C;
|
|
case SDL_SCANCODE_F3: return 0x3D;
|
|
case SDL_SCANCODE_F4: return 0x3E;
|
|
case SDL_SCANCODE_F5: return 0x3F;
|
|
case SDL_SCANCODE_F6: return 0x40;
|
|
case SDL_SCANCODE_F7: return 0x41;
|
|
case SDL_SCANCODE_F8: return 0x42;
|
|
case SDL_SCANCODE_F9: return 0x43;
|
|
case SDL_SCANCODE_F10: return 0x44;
|
|
case SDL_SCANCODE_NUMLOCKCLEAR: return 0x45;
|
|
case SDL_SCANCODE_SCROLLLOCK: return 0x46;
|
|
case SDL_SCANCODE_KP_7: return 0x47;
|
|
case SDL_SCANCODE_KP_8: return 0x48;
|
|
case SDL_SCANCODE_KP_9: return 0x49;
|
|
case SDL_SCANCODE_KP_MINUS: return 0x4A;
|
|
case SDL_SCANCODE_KP_4: return 0x4B;
|
|
case SDL_SCANCODE_KP_5: return 0x4C;
|
|
case SDL_SCANCODE_KP_6: return 0x4D;
|
|
case SDL_SCANCODE_KP_PLUS: return 0x4E;
|
|
case SDL_SCANCODE_KP_1: return 0x4F;
|
|
case SDL_SCANCODE_KP_2: return 0x50;
|
|
case SDL_SCANCODE_KP_3: return 0x51;
|
|
case SDL_SCANCODE_KP_0: return 0x52;
|
|
case SDL_SCANCODE_KP_PERIOD: return 0x53;
|
|
case SDL_SCANCODE_F11: return 0x57;
|
|
case SDL_SCANCODE_F12: return 0x58;
|
|
case SDL_SCANCODE_KP_ENTER: return 0x9C;
|
|
case SDL_SCANCODE_RCTRL: return 0x9D;
|
|
case SDL_SCANCODE_KP_DIVIDE: return 0xB5;
|
|
case SDL_SCANCODE_RALT: return 0xB8;
|
|
case SDL_SCANCODE_HOME: return 0xC7;
|
|
case SDL_SCANCODE_UP: return 0xC8;
|
|
case SDL_SCANCODE_PAGEUP: return 0xC9;
|
|
case SDL_SCANCODE_LEFT: return 0xCB;
|
|
case SDL_SCANCODE_RIGHT: return 0xCD;
|
|
case SDL_SCANCODE_END: return 0xCF;
|
|
case SDL_SCANCODE_DOWN: return 0xD0;
|
|
case SDL_SCANCODE_PAGEDOWN: return 0xD1;
|
|
case SDL_SCANCODE_INSERT: return 0xD2;
|
|
case SDL_SCANCODE_DELETE: return 0xD3;
|
|
default: return 0;
|
|
}
|
|
}
|
|
|
|
static int sdl_scancode_to_vk(SDL_Scancode sc) {
|
|
switch (sc) {
|
|
case SDL_SCANCODE_BACKSPACE: return 0x08;
|
|
case SDL_SCANCODE_TAB: return 0x09;
|
|
case SDL_SCANCODE_RETURN:
|
|
case SDL_SCANCODE_KP_ENTER: return 0x0D;
|
|
case SDL_SCANCODE_ESCAPE: return 0x1B;
|
|
case SDL_SCANCODE_SPACE: return 0x20;
|
|
case SDL_SCANCODE_PAGEUP: return 0x21;
|
|
case SDL_SCANCODE_PAGEDOWN: return 0x22;
|
|
case SDL_SCANCODE_END: return 0x23;
|
|
case SDL_SCANCODE_HOME: return 0x24;
|
|
case SDL_SCANCODE_LEFT: return 0x25;
|
|
case SDL_SCANCODE_UP: return 0x26;
|
|
case SDL_SCANCODE_RIGHT: return 0x27;
|
|
case SDL_SCANCODE_DOWN: return 0x28;
|
|
case SDL_SCANCODE_INSERT: return 0x2D;
|
|
case SDL_SCANCODE_DELETE: return 0x2E;
|
|
case SDL_SCANCODE_F1: return 0x70;
|
|
case SDL_SCANCODE_F2: return 0x71;
|
|
case SDL_SCANCODE_F3: return 0x72;
|
|
case SDL_SCANCODE_F4: return 0x73;
|
|
case SDL_SCANCODE_F5: return 0x74;
|
|
case SDL_SCANCODE_F6: return 0x75;
|
|
case SDL_SCANCODE_F7: return 0x76;
|
|
case SDL_SCANCODE_F8: return 0x77;
|
|
case SDL_SCANCODE_F9: return 0x78;
|
|
case SDL_SCANCODE_F10: return 0x79;
|
|
case SDL_SCANCODE_F11: return 0x7A;
|
|
case SDL_SCANCODE_F12: return 0x7B;
|
|
default: return 0;
|
|
}
|
|
}
|
|
|
|
// Touch hosts: SDL text input (and with it the system keyboard) is on only while the player
|
|
// types into a tapped EditLine; desktop keeps the always-on text input from window creation.
|
|
void sync_screen_keyboard() {
|
|
const bool want = touch_controller.wants_screen_keyboard();
|
|
const unsigned serial = touch_controller.screen_keyboard_serial();
|
|
if (want && (!SDL_TextInputActive(window_) || serial != screen_keyboard_serial_)) {
|
|
SDL_PropertiesID props = SDL_CreateProperties();
|
|
SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_TYPE_NUMBER, SDL_TEXTINPUT_TYPE_TEXT);
|
|
SDL_SetNumberProperty(props, SDL_PROP_TEXTINPUT_CAPITALIZATION_NUMBER, SDL_CAPITALIZE_NONE);
|
|
SDL_SetBooleanProperty(props, SDL_PROP_TEXTINPUT_AUTOCORRECT_BOOLEAN, false);
|
|
SDL_StartTextInputWithProperties(window_, props);
|
|
SDL_DestroyProperties(props);
|
|
} else if (!want && SDL_TextInputActive(window_)) {
|
|
SDL_StopTextInput(window_);
|
|
}
|
|
screen_keyboard_serial_ = serial;
|
|
}
|
|
|
|
bool poll(bool forward_to_live_client = false) {
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
auto map_mouse = [&](float wx, float wy, int& ox, int& oy) {
|
|
int win_w = 0, win_h = 0;
|
|
SDL_GetWindowSize(window_, &win_w, &win_h);
|
|
unsigned ui_w = extent_.width ? extent_.width : 960;
|
|
unsigned ui_h = extent_.height ? extent_.height : 640;
|
|
UIRenderGetSize(&ui_w, &ui_h);
|
|
if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) {
|
|
ox = static_cast<int>(std::lround(wx * float(ui_w) / float(win_w)));
|
|
oy = static_cast<int>(std::lround(wy * float(ui_h) / float(win_h)));
|
|
} else {
|
|
ox = static_cast<int>(std::lround(wx));
|
|
oy = static_cast<int>(std::lround(wy));
|
|
}
|
|
};
|
|
std::optional<std::pair<int, int>> pending_mouse_move;
|
|
auto flush_mouse_move = [&] {
|
|
if (!pending_mouse_move) return;
|
|
const auto [mx, my] = *pending_mouse_move;
|
|
pending_mouse_move.reset();
|
|
last_mx_ = mx;
|
|
last_my_ = my;
|
|
if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return;
|
|
PythonBoot::UIMouseMove(mx, my);
|
|
};
|
|
#endif
|
|
SDL_Event event;
|
|
while (SDL_PollEvent(&event)) {
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
// Touches arrive as SDL_EVENT_FINGER_*; the touch-synthesized mouse copy
|
|
// would register as a second finger (101) and turn every drag into a pinch.
|
|
if (touch_controller.is_enabled() &&
|
|
((event.type == SDL_EVENT_MOUSE_MOTION && event.motion.which == SDL_TOUCH_MOUSEID) ||
|
|
((event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) &&
|
|
event.button.which == SDL_TOUCH_MOUSEID)))
|
|
continue;
|
|
if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) {
|
|
int mx = 0, my = 0;
|
|
map_mouse(event.motion.x, event.motion.y, mx, my);
|
|
pending_mouse_move = std::pair{mx, my};
|
|
continue;
|
|
}
|
|
// Preserve event order at button/key/focus boundaries while collapsing
|
|
// high-frequency motion into the most recent position.
|
|
flush_mouse_move();
|
|
#endif
|
|
if (event.type == SDL_EVENT_QUIT) return false;
|
|
if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) {
|
|
int ew = 0, eh = 0;
|
|
SDL_GetWindowSize(window_, &ew, &eh);
|
|
touch_controller.update_screen_size(int(logical_width()), int(logical_height()), ui_safe_inset());
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
if (forward_to_live_client && ew > 0 && eh > 0) {
|
|
PythonBoot::SetUISafeInset(ui_safe_inset());
|
|
PythonBoot::SetUISize(int(logical_width()), int(logical_height()));
|
|
}
|
|
#endif
|
|
recreate_swapchain();
|
|
continue;
|
|
}
|
|
if (event.type == SDL_EVENT_FINGER_DOWN) {
|
|
touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y);
|
|
continue;
|
|
} else if (event.type == SDL_EVENT_FINGER_MOTION) {
|
|
touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y);
|
|
continue;
|
|
} else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) {
|
|
touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y);
|
|
continue;
|
|
}
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
if (forward_to_live_client) {
|
|
if (!hardware_cursor_enabled_ &&
|
|
(event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) {
|
|
std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n");
|
|
SDL_HideCursor();
|
|
os_cursor_hidden_ = true;
|
|
} else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) {
|
|
SDL_CaptureMouse(false);
|
|
PythonBoot::UIMouseButton(2, false, last_mx_, last_my_);
|
|
PythonBoot::UIMouseButton(3, false, last_mx_, last_my_);
|
|
PythonBoot::UIMouseButton(1, false, last_mx_, last_my_);
|
|
}
|
|
if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) {
|
|
const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN;
|
|
int btn = 0;
|
|
if (event.button.button == SDL_BUTTON_LEFT) btn = 1;
|
|
else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2;
|
|
else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3;
|
|
if (btn) {
|
|
int mx = 0, my = 0;
|
|
map_mouse(event.button.x, event.button.y, mx, my);
|
|
last_mx_ = mx;
|
|
last_my_ = my;
|
|
if (touch_controller.is_enabled()) {
|
|
if (touch_controller.on_mouse_button(btn, pressed, mx, my)) {
|
|
continue;
|
|
}
|
|
}
|
|
SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0);
|
|
PythonBoot::UIMouseButton(btn, pressed, mx, my);
|
|
}
|
|
} else if (event.type == SDL_EVENT_MOUSE_WHEEL) {
|
|
PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f));
|
|
} else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) {
|
|
const bool pressed = event.type == SDL_EVENT_KEY_DOWN;
|
|
// Android back = ESC: closes the top window, or opens the system menu (game.py).
|
|
if (event.key.scancode == SDL_SCANCODE_AC_BACK) event.key.scancode = SDL_SCANCODE_ESCAPE;
|
|
if (pressed) {
|
|
if (const int vk = sdl_scancode_to_vk(event.key.scancode))
|
|
PythonBoot::UIIMEKeyDown(vk);
|
|
if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8);
|
|
else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9);
|
|
else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER)
|
|
PythonBoot::UIChar(13);
|
|
else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27);
|
|
}
|
|
if (!event.key.repeat) {
|
|
if (const int dik = sdl_scancode_to_dik(event.key.scancode))
|
|
PythonBoot::UIKey(dik, pressed);
|
|
}
|
|
} else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) {
|
|
const auto* s = reinterpret_cast<const unsigned char*>(event.text.text);
|
|
while (*s) {
|
|
unsigned cp = 0;
|
|
if (*s < 0x80u) {
|
|
cp = *s++;
|
|
} else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) {
|
|
cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu);
|
|
s += 2;
|
|
} else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) {
|
|
cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu);
|
|
s += 3;
|
|
} else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) {
|
|
cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu);
|
|
s += 4;
|
|
} else {
|
|
++s;
|
|
continue;
|
|
}
|
|
if (cp >= 32u && cp != 127u)
|
|
PythonBoot::UIChar(cp);
|
|
}
|
|
}
|
|
}
|
|
#else
|
|
(void)forward_to_live_client;
|
|
#endif
|
|
}
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
flush_mouse_move();
|
|
#endif
|
|
touch_controller.update();
|
|
if (touch_controller.is_enabled())
|
|
sync_screen_keyboard();
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) {
|
|
SDL_HideCursor();
|
|
os_cursor_hidden_ = true;
|
|
}
|
|
#endif
|
|
return true;
|
|
}
|
|
|
|
void reset_timings() {
|
|
finish_gpu_timings();
|
|
timings_ = {};
|
|
timed_frames_ = 0;
|
|
upload_count_ = 0;
|
|
uploaded_bytes_ = 0;
|
|
texture_upload_count_ = 0;
|
|
texture_uploaded_bytes_ = 0;
|
|
}
|
|
|
|
void finish_gpu_timings() {
|
|
for (auto& slot : frames_) {
|
|
if (!slot.has_pending_query) continue;
|
|
check(vkWaitForFences(device_, 1, &slot.fence, VK_TRUE, UINT64_MAX), "vkWaitForFences finish");
|
|
collect_pending_gpu_timestamp(slot);
|
|
}
|
|
}
|
|
|
|
// 1 = the old serial loop (CPU waits for the previous frame's GPU work), 2 = CPU/GPU overlap.
|
|
void set_frames_in_flight(std::uint32_t count) {
|
|
frames_in_flight_ = std::clamp<std::uint32_t>(count, 1, kMaxFramesInFlight);
|
|
}
|
|
std::uint32_t frames_in_flight() const { return frames_in_flight_; }
|
|
|
|
void render(
|
|
const std::vector<Render3DDraw>& draws,
|
|
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures,
|
|
std::uint32_t ui_width = 960,
|
|
std::uint32_t ui_height = 640,
|
|
const std::vector<UIRenderCommand>& ui_commands = {}) {
|
|
if (extent_.width == 0 || extent_.height == 0) return;
|
|
const auto start = std::chrono::steady_clock::now();
|
|
f_ = &frames_[frame_slot_ % frames_in_flight_];
|
|
check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences");
|
|
collect_pending_gpu_timestamp(*f_);
|
|
reset_staging(*f_);
|
|
const auto fenced = std::chrono::steady_clock::now();
|
|
prune_textures();
|
|
std::uint32_t image_index = 0;
|
|
const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, f_->acquire, VK_NULL_HANDLE, &image_index);
|
|
if (acquired == VK_ERROR_OUT_OF_DATE_KHR) {
|
|
recreate_swapchain();
|
|
return;
|
|
}
|
|
if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR");
|
|
const auto synchronized = std::chrono::steady_clock::now();
|
|
++frame_number_;
|
|
++timed_frames_;
|
|
while (rendered_.size() < swapchain_images_.size()) {
|
|
VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO};
|
|
VkSemaphore semaphore = VK_NULL_HANDLE;
|
|
check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &semaphore), "vkCreateSemaphore rendered");
|
|
rendered_.push_back(semaphore);
|
|
}
|
|
// Texture uploads of this frame are recorded ahead of the render pass (upload_texture).
|
|
check(vkResetCommandBuffer(f_->command, 0), "vkResetCommandBuffer");
|
|
VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
|
|
begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
|
|
check(vkBeginCommandBuffer(f_->command, &begin), "vkBeginCommandBuffer");
|
|
vkCmdResetQueryPool(f_->command, query_pool_, f_->query_base, 2);
|
|
vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, f_->query_base);
|
|
f_->recording = true;
|
|
|
|
std::vector<PreparedDraw> prepared_draws;
|
|
prepared_draws.reserve(draws.size());
|
|
// Draws into offscreen render targets (the shadow map), in recorded order, drawn before the
|
|
// back-buffer pass that samples them.
|
|
std::vector<PreparedDraw> offscreen_draws;
|
|
std::unordered_set<std::uint64_t> active_keys;
|
|
std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0;
|
|
std::uint32_t bone_cursor = 0;
|
|
std::size_t required_bones = 0;
|
|
for (const auto& draw : draws) {
|
|
if (!draw.bone_matrices.empty()) {
|
|
if (draw.bone_matrices.size() % 16)
|
|
throw std::runtime_error("invalid bone palette length");
|
|
required_bones += draw.bone_matrices.size() / 16;
|
|
}
|
|
}
|
|
ensure_bone_capacity(required_bones);
|
|
|
|
for (const auto& draw : draws) {
|
|
RenderTarget* target = nullptr;
|
|
if (draw.render_target) {
|
|
target = get_render_target(
|
|
Render3DRenderTargetName(draw.render_target, draw.target_width, draw.target_height));
|
|
if (!target) continue;
|
|
}
|
|
if (draw.clear_flags && target) {
|
|
PreparedDraw clear{};
|
|
clear.target = target;
|
|
clear.clear_flags = draw.clear_flags;
|
|
const auto color = unpack_argb(draw.clear_color);
|
|
clear.clear_color.color = {{color[0], color[1], color[2], color[3]}};
|
|
clear.clear_depth.depthStencil = {draw.clear_z, 0};
|
|
clear.clear_rect = {{0, 0}, {target->width, target->height}};
|
|
offscreen_draws.push_back(clear);
|
|
continue;
|
|
}
|
|
if (draw.clear_flags) {
|
|
PreparedDraw clear{};
|
|
clear.clear_flags = draw.clear_flags;
|
|
const auto color = unpack_argb(draw.clear_color);
|
|
clear.clear_color.color = {{color[0], color[1], color[2], color[3]}};
|
|
clear.clear_depth.depthStencil = {draw.clear_z, 0};
|
|
// D3D8 Clear with no rectangles clears the current viewport.
|
|
const auto viewport = draw_viewport(draw);
|
|
const float x0 = std::clamp(viewport.x, 0.0f, float(extent_.width));
|
|
const float y0 = std::clamp(viewport.y, 0.0f, float(extent_.height));
|
|
const float x1 = std::clamp(viewport.x + viewport.width, 0.0f, float(extent_.width));
|
|
const float y1 = std::clamp(viewport.y + viewport.height, 0.0f, float(extent_.height));
|
|
clear.clear_rect = {{std::int32_t(x0), std::int32_t(y0)},
|
|
{std::uint32_t(x1 - x0), std::uint32_t(y1 - y0)}};
|
|
prepared_draws.push_back(clear);
|
|
continue;
|
|
}
|
|
if (draw.positions.empty() || draw.indices.empty()) continue;
|
|
const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw);
|
|
const GeometryId id{draw.geometry_key, signature};
|
|
auto [it, inserted] = geometries_.try_emplace(id);
|
|
auto& geometry = it->second;
|
|
if (inserted) {
|
|
upload_geometry(draw, geometry);
|
|
} else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) {
|
|
throw std::runtime_error("geometry key/revision reused with a different size");
|
|
}
|
|
geometry.last_used_frame = frame_number_;
|
|
if (draw.geometry_key) active_keys.insert(draw.geometry_key);
|
|
|
|
float skin_offset_encoded = 0.0f;
|
|
if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && f_->bone_mapped) {
|
|
const auto bone_count = static_cast<std::uint32_t>(draw.bone_matrices.size() / 16);
|
|
if (std::size_t(bone_cursor) + bone_count <= f_->bone_capacity) {
|
|
std::memcpy(
|
|
f_->bone_mapped + std::size_t(bone_cursor) * 16,
|
|
draw.bone_matrices.data(),
|
|
draw.bone_matrices.size() * sizeof(float));
|
|
skin_offset_encoded = float(bone_cursor + 1u);
|
|
bone_cursor += bone_count;
|
|
++skinned_draw_count;
|
|
} else throw std::runtime_error("bone palette capacity exceeded");
|
|
}
|
|
|
|
const bool is_alpha = draw.alpha_blend != 0;
|
|
std::uint8_t cull = 0;
|
|
// D3D8 cull mode names describe the winding to discard. With the
|
|
// shader's Y flip, clockwise is Vulkan's front-facing winding.
|
|
if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise
|
|
else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise
|
|
const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0;
|
|
std::uint8_t blend = 0;
|
|
if (is_alpha) {
|
|
if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 &&
|
|
draw.dest_blend >= 1 && draw.dest_blend <= 11) {
|
|
blend = static_cast<std::uint8_t>((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4));
|
|
} else {
|
|
blend = 1;
|
|
}
|
|
}
|
|
const auto pipeline = get_pipeline(cull, depth, blend, draw.lines,
|
|
static_cast<std::uint8_t>(draw.z_func), target != nullptr);
|
|
// D3DTOP_DISABLE (1) on stage 0 or 1 ends the cascade before stage 1 samples.
|
|
const bool bind_tex1 = !draw.texture1.empty() && draw.color_op[0] > 1 && draw.color_op[1] > 1;
|
|
const auto descriptor = get_texture_descriptor(
|
|
draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw);
|
|
|
|
auto constants = make_push_constants(draw, skin_offset_encoded);
|
|
if (draw.pretransformed) {
|
|
const float screen_width = float(logical_width());
|
|
const float screen_height = float(logical_height());
|
|
// Screen pixels (y down) to D3D NDC (y up); the shader's Y flip then maps to Vulkan.
|
|
constants.mvp = {2.0f / screen_width, 0, 0, 0,
|
|
0, -2.0f / screen_height, 0, 0,
|
|
0, 0, 1, 0,
|
|
-1, 1, 0, 1};
|
|
}
|
|
PreparedDraw prepared_draw{};
|
|
prepared_draw.target = target;
|
|
prepared_draw.geometry = &geometry;
|
|
prepared_draw.pipeline = pipeline;
|
|
prepared_draw.descriptor = descriptor;
|
|
prepared_draw.constants = constants;
|
|
prepared_draw.fixed_state = make_fixed_function_state(draw);
|
|
// XYZRHW vertices are already in screen space and bypass the viewport transform.
|
|
prepared_draw.viewport = draw.pretransformed ? full_viewport() : draw_viewport(draw);
|
|
if (draw.pretransformed && draw.ui_offset_x != 0)
|
|
prepared_draw.viewport.x += draw.ui_offset_x * float(extent_.width) / float(logical_width());
|
|
if (target) {
|
|
// D3DVIEWPORT8 in the target's own pixels.
|
|
prepared_draw.viewport = draw.viewport[2] > 0 && draw.viewport[3] > 0
|
|
? VkViewport{draw.viewport[0], draw.viewport[1], draw.viewport[2], draw.viewport[3],
|
|
draw.viewport_z[0], draw.viewport_z[1]}
|
|
: VkViewport{0, 0, float(target->width), float(target->height), 0, 1};
|
|
offscreen_draws.push_back(prepared_draw);
|
|
} else prepared_draws.push_back(prepared_draw);
|
|
vertex_count += geometry.vertex_count;
|
|
index_count += geometry.index_count;
|
|
}
|
|
for (auto it = geometries_.begin(); it != geometries_.end();) {
|
|
const auto age = frame_number_ - it->second.last_used_frame;
|
|
const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key);
|
|
if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) {
|
|
release_geometry(it->second);
|
|
it = geometries_.erase(it);
|
|
} else ++it;
|
|
}
|
|
|
|
std::vector<UiBatch> ui_batches;
|
|
std::size_t ui_quad_count = 0;
|
|
if (f_->ui_mapped) {
|
|
const uint32_t target_ui_w = ui_width ? ui_width : 960;
|
|
const uint32_t target_ui_h = ui_height ? ui_height : 640;
|
|
update_fps_counter();
|
|
if (touch_controller.is_enabled() || show_fps_) {
|
|
std::vector<UIRenderCommand> combined_ui = ui_commands;
|
|
if (touch_controller.is_enabled()) {
|
|
touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h), ui_safe_inset());
|
|
touch_controller.append_ui_commands(combined_ui);
|
|
}
|
|
if (show_fps_ && fps_ >= 0) {
|
|
// Top left, inside the safe area and clear of 40250's corner icon.
|
|
const float inset = float(std::min<int>(ui_safe_inset(), int(target_ui_w) / 8));
|
|
TouchController::draw_segment_text(combined_ui, "FPS " + std::to_string(fps_),
|
|
inset + 40.0f, 8.0f, 12.0f, 0xFF40FF40);
|
|
}
|
|
if (!combined_ui.empty()) {
|
|
build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count);
|
|
}
|
|
} else if (!ui_commands.empty()) {
|
|
build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count);
|
|
}
|
|
}
|
|
|
|
ensure_state_capacity(prepared_draws.size() + offscreen_draws.size() + ui_batches.size());
|
|
std::size_t state_index = 0;
|
|
for (auto& draw : offscreen_draws) {
|
|
if (!draw.geometry) continue;
|
|
draw.state_offset = static_cast<std::uint32_t>(state_index * state_stride_);
|
|
std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState));
|
|
++state_index;
|
|
}
|
|
last_offscreen_draw_count_ = offscreen_draws.size();
|
|
for (auto& draw : prepared_draws) {
|
|
if (!draw.geometry) continue;
|
|
draw.state_offset = static_cast<std::uint32_t>(state_index * state_stride_);
|
|
std::memcpy(f_->state_mapped + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState));
|
|
++state_index;
|
|
}
|
|
const FixedFunctionState ui_state{};
|
|
for (auto& batch : ui_batches) {
|
|
batch.state_offset = static_cast<std::uint32_t>(state_index * state_stride_);
|
|
std::memcpy(f_->state_mapped + batch.state_offset, &ui_state, sizeof(FixedFunctionState));
|
|
++state_index;
|
|
}
|
|
|
|
const auto prepared = std::chrono::steady_clock::now();
|
|
f_->recording = false;
|
|
check(vkResetFences(device_, 1, &f_->fence), "vkResetFences");
|
|
|
|
VkClearValue clears[3]{};
|
|
// 40250 clears the back buffer to 0xff000000 once at device creation and afterwards
|
|
// only clears depth each frame (CPythonApplication::Process -> ClearDepthBuffer).
|
|
clears[0].color = {{0.0f, 0.0f, 0.0f, 1.0f}};
|
|
clears[1].depthStencil = {1.0f, 0};
|
|
VkPipeline current_pipeline = VK_NULL_HANDLE;
|
|
VkDescriptorSet current_descriptor = VK_NULL_HANDLE;
|
|
|
|
VkViewport current_viewport{-1, -1, -1, -1, -1, -1};
|
|
auto set_viewport = [&](const VkViewport& viewport) {
|
|
if (std::memcmp(&viewport, ¤t_viewport, sizeof(viewport)) == 0) return;
|
|
vkCmdSetViewport(f_->command, 0, 1, &viewport);
|
|
current_viewport = viewport;
|
|
};
|
|
|
|
// One prepared draw or Clear inside the current render pass.
|
|
auto record_prepared = [&](const PreparedDraw& prepared_draw) {
|
|
if (!prepared_draw.geometry) {
|
|
VkClearAttachment attachments[2]{};
|
|
std::uint32_t attachment_count = 0;
|
|
if (prepared_draw.clear_flags & 1u) { // D3DCLEAR_TARGET
|
|
attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
|
attachments[attachment_count].colorAttachment = 0;
|
|
attachments[attachment_count].clearValue = prepared_draw.clear_color;
|
|
++attachment_count;
|
|
}
|
|
if (prepared_draw.clear_flags & 2u) { // D3DCLEAR_ZBUFFER
|
|
attachments[attachment_count].aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
|
attachments[attachment_count].clearValue = prepared_draw.clear_depth;
|
|
++attachment_count;
|
|
}
|
|
const VkClearRect rect{prepared_draw.clear_rect, 0, 1};
|
|
if (attachment_count && rect.rect.extent.width && rect.rect.extent.height)
|
|
vkCmdClearAttachments(f_->command, attachment_count, attachments, 1, &rect);
|
|
return;
|
|
}
|
|
set_viewport(prepared_draw.viewport);
|
|
if (prepared_draw.pipeline != current_pipeline) {
|
|
vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline);
|
|
current_pipeline = prepared_draw.pipeline;
|
|
}
|
|
if (prepared_draw.descriptor != current_descriptor) {
|
|
vkCmdBindDescriptorSets(
|
|
f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr);
|
|
current_descriptor = prepared_draw.descriptor;
|
|
}
|
|
vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1,
|
|
&f_->bone_descriptor_set, 1, &prepared_draw.state_offset);
|
|
const auto& geometry = *prepared_draw.geometry;
|
|
const VkDeviceSize vertex_offset = 0;
|
|
vkCmdBindVertexBuffers(f_->command, 0, 1, &geometry.buffer, &vertex_offset);
|
|
vkCmdBindIndexBuffer(f_->command, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32);
|
|
vkCmdPushConstants(
|
|
f_->command,
|
|
layout_,
|
|
VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT,
|
|
0,
|
|
sizeof(prepared_draw.constants),
|
|
&prepared_draw.constants);
|
|
vkCmdDrawIndexed(f_->command, geometry.index_count, 1, 0, 0, 0);
|
|
};
|
|
|
|
// Render targets first used this frame (drawn or only sampled) start cleared.
|
|
for (auto& entry : render_targets_)
|
|
if (!entry.second.initialized) initialize_render_target(entry.second);
|
|
// Offscreen passes: one per run of draws into the same target.
|
|
for (std::size_t i = 0; i < offscreen_draws.size();) {
|
|
RenderTarget* target = offscreen_draws[i].target;
|
|
VkRenderPassBeginInfo offscreen_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO};
|
|
offscreen_begin.renderPass = offscreen_pass_;
|
|
offscreen_begin.framebuffer = target->framebuffer;
|
|
offscreen_begin.renderArea.extent = {target->width, target->height};
|
|
vkCmdBeginRenderPass(f_->command, &offscreen_begin, VK_SUBPASS_CONTENTS_INLINE);
|
|
for (; i < offscreen_draws.size() && offscreen_draws[i].target == target; ++i)
|
|
record_prepared(offscreen_draws[i]);
|
|
vkCmdEndRenderPass(f_->command);
|
|
}
|
|
|
|
VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO};
|
|
pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index);
|
|
pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = samples_ != VK_SAMPLE_COUNT_1_BIT ? 3 : 2; pass_begin.pClearValues = clears;
|
|
vkCmdBeginRenderPass(f_->command, &pass_begin, VK_SUBPASS_CONTENTS_INLINE);
|
|
|
|
auto record_ui_pass = [&](bool behind_3d) {
|
|
bool ui_bound = false;
|
|
set_viewport(full_viewport());
|
|
for (const auto& batch : ui_batches) {
|
|
if (batch.behind_3d != behind_3d) continue;
|
|
if (!ui_bound) {
|
|
const VkDeviceSize v_offset = 0;
|
|
vkCmdBindVertexBuffers(f_->command, 0, 1, &f_->ui_buffer, &v_offset);
|
|
vkCmdBindIndexBuffer(f_->command, f_->ui_buffer, f_->ui_vertex_bytes, VK_INDEX_TYPE_UINT32);
|
|
ui_bound = true;
|
|
}
|
|
if (batch.pipeline != current_pipeline) {
|
|
vkCmdBindPipeline(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline);
|
|
current_pipeline = batch.pipeline;
|
|
}
|
|
if (batch.descriptor != current_descriptor) {
|
|
vkCmdBindDescriptorSets(
|
|
f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr);
|
|
current_descriptor = batch.descriptor;
|
|
}
|
|
vkCmdBindDescriptorSets(f_->command, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1,
|
|
&f_->bone_descriptor_set, 1, &batch.state_offset);
|
|
vkCmdPushConstants(
|
|
f_->command,
|
|
layout_,
|
|
VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT,
|
|
0,
|
|
sizeof(batch.constants),
|
|
&batch.constants);
|
|
vkCmdDrawIndexed(f_->command, batch.index_count, 1, batch.first_index, 0, 0);
|
|
}
|
|
};
|
|
|
|
record_ui_pass(true);
|
|
|
|
for (const auto& prepared_draw : prepared_draws) record_prepared(prepared_draw);
|
|
|
|
record_ui_pass(false);
|
|
|
|
vkCmdEndRenderPass(f_->command);
|
|
VkBuffer readback = VK_NULL_HANDLE;
|
|
VkDeviceMemory readback_memory = VK_NULL_HANDLE;
|
|
const bool screenshot = !screenshot_path_.empty() && frame_number_ == screenshot_frame_;
|
|
if (screenshot) {
|
|
const VkDeviceSize bytes = VkDeviceSize(extent_.width) * extent_.height * 4;
|
|
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
|
|
info.size = bytes; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
|
|
check(vkCreateBuffer(device_, &info, nullptr, &readback), "vkCreateBuffer readback");
|
|
VkMemoryRequirements requirements{};
|
|
vkGetBufferMemoryRequirements(device_, readback, &requirements);
|
|
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
allocation.allocationSize = requirements.size;
|
|
allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits,
|
|
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
|
|
check(vkAllocateMemory(device_, &allocation, nullptr, &readback_memory), "vkAllocateMemory readback");
|
|
check(vkBindBufferMemory(device_, readback, readback_memory, 0), "vkBindBufferMemory readback");
|
|
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
|
|
barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
|
|
barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
|
|
barrier.oldLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
|
|
barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
|
|
barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
barrier.image = swapchain_images_.at(image_index);
|
|
barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
|
|
vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT,
|
|
VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier);
|
|
VkBufferImageCopy region{};
|
|
region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1};
|
|
region.imageExtent = {extent_.width, extent_.height, 1};
|
|
vkCmdCopyImageToBuffer(f_->command, barrier.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, readback, 1, ®ion);
|
|
barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
|
|
barrier.dstAccessMask = 0;
|
|
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
|
|
barrier.newLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
|
|
vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT,
|
|
VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier);
|
|
}
|
|
// Debug: MT_DUMP_RT=prefix writes each render target of the screenshot frame to <prefix>_<n>.pgm.
|
|
struct RtDump { VkBuffer buffer; VkDeviceMemory memory; std::uint32_t w, h; };
|
|
std::vector<RtDump> rt_dumps;
|
|
const char* dump_rt = std::getenv("MT_DUMP_RT");
|
|
if (screenshot && dump_rt && *dump_rt) {
|
|
const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4;
|
|
for (auto& entry : render_targets_) {
|
|
RenderTarget& rt = entry.second;
|
|
RtDump dump{VK_NULL_HANDLE, VK_NULL_HANDLE, rt.width, rt.height};
|
|
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
|
|
info.size = VkDeviceSize(rt.width) * rt.height * texel; info.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT;
|
|
check(vkCreateBuffer(device_, &info, nullptr, &dump.buffer), "vkCreateBuffer rt dump");
|
|
VkMemoryRequirements requirements{};
|
|
vkGetBufferMemoryRequirements(device_, dump.buffer, &requirements);
|
|
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
allocation.allocationSize = requirements.size;
|
|
allocation.memoryTypeIndex = find_memory_type(requirements.memoryTypeBits,
|
|
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
|
|
check(vkAllocateMemory(device_, &allocation, nullptr, &dump.memory), "vkAllocateMemory rt dump");
|
|
check(vkBindBufferMemory(device_, dump.buffer, dump.memory, 0), "vkBindBufferMemory rt dump");
|
|
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
|
|
barrier.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_SHADER_READ_BIT;
|
|
barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
|
|
barrier.oldLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
|
|
barrier.srcQueueFamilyIndex = barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
barrier.image = rt.texture.image;
|
|
barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
|
|
vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT,
|
|
VK_PIPELINE_STAGE_TRANSFER_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier);
|
|
VkBufferImageCopy region{};
|
|
region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1};
|
|
region.imageExtent = {rt.width, rt.height, 1};
|
|
vkCmdCopyImageToBuffer(f_->command, rt.texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dump.buffer, 1, ®ion);
|
|
barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
|
|
barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
|
|
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
|
|
barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT,
|
|
VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, nullptr, 0, nullptr, 1, &barrier);
|
|
rt_dumps.push_back(dump);
|
|
}
|
|
}
|
|
vkCmdWriteTimestamp(f_->command, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, f_->query_base + 1);
|
|
check(vkEndCommandBuffer(f_->command), "vkEndCommandBuffer");
|
|
f_->has_pending_query = true;
|
|
|
|
const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT;
|
|
VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO};
|
|
submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &f_->acquire; submit.pWaitDstStageMask = &stage;
|
|
submit.commandBufferCount = 1; submit.pCommandBuffers = &f_->command;
|
|
submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_[image_index];
|
|
check(vkQueueSubmit(queue_, 1, &submit, f_->fence), "vkQueueSubmit");
|
|
++frame_slot_;
|
|
if (screenshot) {
|
|
check(vkWaitForFences(device_, 1, &f_->fence, VK_TRUE, UINT64_MAX), "vkWaitForFences readback");
|
|
void* mapped = nullptr;
|
|
check(vkMapMemory(device_, readback_memory, 0, VK_WHOLE_SIZE, 0, &mapped), "vkMapMemory readback");
|
|
write_bmp(screenshot_path_, static_cast<const std::uint8_t*>(mapped));
|
|
vkUnmapMemory(device_, readback_memory);
|
|
const std::uint32_t texel = offscreen_format_ == VK_FORMAT_R5G6B5_UNORM_PACK16 ? 2 : 4;
|
|
for (std::size_t n = 0; n < rt_dumps.size(); ++n) {
|
|
const auto& dump = rt_dumps[n];
|
|
void* data = nullptr;
|
|
check(vkMapMemory(device_, dump.memory, 0, VK_WHOLE_SIZE, 0, &data), "vkMapMemory rt dump");
|
|
const auto* bytes = static_cast<const std::uint8_t*>(data);
|
|
std::ofstream out(std::string(dump_rt) + "_" + std::to_string(n) + ".pgm", std::ios::binary);
|
|
out << "P5\n" << dump.w << " " << dump.h << "\n255\n";
|
|
for (std::size_t i = 0; i < std::size_t(dump.w) * dump.h; ++i) {
|
|
std::uint8_t g = texel == 2 ? std::uint8_t(((bytes[i * 2 + 1] >> 3) & 31) * 255 / 31) : bytes[i * 4];
|
|
out.put(char(g));
|
|
}
|
|
vkUnmapMemory(device_, dump.memory);
|
|
vkDestroyBuffer(device_, dump.buffer, nullptr);
|
|
vkFreeMemory(device_, dump.memory, nullptr);
|
|
}
|
|
vkDestroyBuffer(device_, readback, nullptr);
|
|
vkFreeMemory(device_, readback_memory, nullptr);
|
|
}
|
|
const auto submitted = std::chrono::steady_clock::now();
|
|
VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR};
|
|
present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_[image_index];
|
|
present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index;
|
|
const auto presented = vkQueuePresentKHR(queue_, &present);
|
|
// An Android surface can continuously report SUBOPTIMAL while the app deliberately
|
|
// uses the identity transform. The image is still presentable; rebuilding every frame
|
|
// stalls rendering without changing that status.
|
|
if (presented == VK_ERROR_OUT_OF_DATE_KHR) {
|
|
recreate_swapchain();
|
|
} else if (presented != VK_SUCCESS && presented != VK_SUBOPTIMAL_KHR) {
|
|
check(presented, "vkQueuePresentKHR");
|
|
}
|
|
const auto presented_at = std::chrono::steady_clock::now();
|
|
const auto prepare_ms = std::chrono::duration<double, std::milli>(prepared - synchronized).count();
|
|
last_sync_ms_ = std::chrono::duration<double, std::milli>(synchronized - start).count();
|
|
timings_.sync_ms += last_sync_ms_;
|
|
timings_.fence_ms += std::chrono::duration<double, std::milli>(fenced - start).count();
|
|
timings_.prepare_ms += prepare_ms;
|
|
if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms;
|
|
timings_.submit_ms += std::chrono::duration<double, std::milli>(submitted - prepared).count();
|
|
timings_.present_ms += std::chrono::duration<double, std::milli>(presented_at - submitted).count();
|
|
last_draw_count_ = prepared_draws.size();
|
|
last_skinned_draw_count_ = skinned_draw_count;
|
|
last_ui_batch_count_ = ui_batches.size();
|
|
last_ui_quad_count_ = ui_quad_count;
|
|
last_vertex_count_ = vertex_count;
|
|
last_index_count_ = index_count;
|
|
}
|
|
|
|
std::size_t draw_count() const { return last_draw_count_; }
|
|
std::size_t skinned_draw_count() const { return last_skinned_draw_count_; }
|
|
std::size_t ui_batch_count() const { return last_ui_batch_count_; }
|
|
std::size_t ui_quad_count() const { return last_ui_quad_count_; }
|
|
std::size_t vertex_count() const { return last_vertex_count_; }
|
|
std::size_t index_count() const { return last_index_count_; }
|
|
std::size_t upload_count() const { return upload_count_; }
|
|
std::size_t uploaded_bytes() const { return uploaded_bytes_; }
|
|
std::size_t texture_upload_count() const { return texture_upload_count_; }
|
|
std::size_t texture_uploaded_bytes() const { return texture_uploaded_bytes_; }
|
|
const std::string& device_name() const { return device_name_; }
|
|
const char* present_mode_name() const {
|
|
switch (present_mode_) {
|
|
case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE";
|
|
case VK_PRESENT_MODE_MAILBOX_KHR: return "MAILBOX";
|
|
default: return "FIFO";
|
|
}
|
|
}
|
|
Timings timings() const { return timings_; }
|
|
native_perf::RendererTotals perf_totals() const {
|
|
native_perf::RendererTotals t;
|
|
t.sync_ms = timings_.sync_ms;
|
|
t.fence_ms = timings_.fence_ms;
|
|
t.prepare_ms = timings_.prepare_ms;
|
|
t.geometry_upload_ms = timings_.geometry_upload_ms;
|
|
t.texture_upload_ms = timings_.texture_upload_ms;
|
|
t.submit_ms = timings_.submit_ms;
|
|
t.present_ms = timings_.present_ms;
|
|
t.gpu_ms = timings_.gpu_ms;
|
|
t.gpu_samples = timings_.gpu_samples;
|
|
t.uploads = upload_count_;
|
|
t.uploaded_bytes = uploaded_bytes_;
|
|
t.texture_uploads = texture_upload_count_;
|
|
t.texture_bytes = texture_uploaded_bytes_;
|
|
t.draws = last_draw_count_;
|
|
t.skinned_draws = last_skinned_draw_count_;
|
|
t.ui_batches = last_ui_batch_count_;
|
|
t.vertices = last_vertex_count_;
|
|
t.last_texture = &last_texture_upload_name_;
|
|
return t;
|
|
}
|
|
|
|
private:
|
|
void recreate_swapchain() {
|
|
int w = 0, h = 0;
|
|
SDL_GetWindowSizeInPixels(window_, &w, &h);
|
|
if (w <= 0 || h <= 0) return;
|
|
vkDeviceWaitIdle(device_);
|
|
for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr);
|
|
pipelines_.clear();
|
|
for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr);
|
|
framebuffers_.clear();
|
|
if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; }
|
|
if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; }
|
|
if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; }
|
|
destroy_msaa_color();
|
|
for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr);
|
|
image_views_.clear();
|
|
if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; }
|
|
create_swapchain();
|
|
create_framebuffers();
|
|
get_pipeline(0, 2, 0);
|
|
}
|
|
|
|
void collect_pending_gpu_timestamp(FrameSlot& slot) {
|
|
if (!slot.has_pending_query) return;
|
|
slot.has_pending_query = false;
|
|
std::uint64_t timestamps[2] = {};
|
|
const VkResult res = vkGetQueryPoolResults(
|
|
device_, query_pool_, slot.query_base, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t),
|
|
VK_QUERY_RESULT_64_BIT);
|
|
if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) {
|
|
const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_);
|
|
timings_.gpu_ms += ns * 1e-6;
|
|
++timings_.gpu_samples;
|
|
}
|
|
}
|
|
|
|
void build_ui_batches(
|
|
std::uint32_t ui_width,
|
|
std::uint32_t ui_height,
|
|
const std::vector<UIRenderCommand>& commands,
|
|
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures,
|
|
std::vector<UiBatch>& batches,
|
|
std::size_t& out_quad_count) {
|
|
ensure_ui_capacity(commands.size());
|
|
auto* vertices = reinterpret_cast<Vertex*>(f_->ui_mapped);
|
|
auto* indices = reinterpret_cast<std::uint32_t*>(f_->ui_mapped + f_->ui_vertex_bytes);
|
|
std::uint32_t v_count = 0;
|
|
std::uint32_t i_count = 0;
|
|
const float inv_w = 2.0f / float(ui_width);
|
|
const float inv_h = 2.0f / float(ui_height);
|
|
|
|
for (const auto& cmd : commands) {
|
|
if (v_count + 4 > f_->ui_vertex_capacity || i_count + 6 > f_->ui_index_capacity)
|
|
throw std::runtime_error("UI command buffer capacity exceeded");
|
|
if (cmd.kind == UIRenderCommand::Text) continue;
|
|
|
|
float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2;
|
|
float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev;
|
|
const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1;
|
|
|
|
float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {};
|
|
std::array<float, 4> top_color = unpack_argb(cmd.argb);
|
|
std::array<float, 4> bot_color =
|
|
cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color;
|
|
if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue;
|
|
|
|
if (cmd.kind == UIRenderCommand::Line) {
|
|
const float dx = x2 - x1, dy = y2 - y1;
|
|
const float len = std::sqrt(dx * dx + dy * dy);
|
|
if (len < 0.1f) continue;
|
|
const float nx = -dy / len * 0.5f;
|
|
const float ny = dx / len * 0.5f;
|
|
qx[0] = x1 + nx; qy[0] = y1 + ny;
|
|
qx[1] = x2 + nx; qy[1] = y2 + ny;
|
|
qx[2] = x1 - nx; qy[2] = y1 - ny;
|
|
qx[3] = x2 - nx; qy[3] = y2 - ny;
|
|
for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; }
|
|
} else if (cmd.kind == UIRenderCommand::Image && cmd.quad) {
|
|
const bool axis_aligned =
|
|
std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f &&
|
|
std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f &&
|
|
std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f &&
|
|
std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f;
|
|
for (int k = 0; k < 4; ++k) {
|
|
mu[k] = cmd.mu[k];
|
|
mv[k] = cmd.mv[k];
|
|
}
|
|
if (has_clip && axis_aligned) {
|
|
const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3];
|
|
const float cx1 = std::max(ox1, cmd.clip_x1);
|
|
const float cy1 = std::max(oy1, cmd.clip_y1);
|
|
const float cx2 = std::min(ox2, cmd.clip_x2);
|
|
const float cy2 = std::min(oy2, cmd.clip_y2);
|
|
if (cx2 <= cx1 || cy2 <= cy1) continue;
|
|
float msu = cmd.mu[0], meu = cmd.mu[1];
|
|
float msv = cmd.mv[0], mev = cmd.mv[2];
|
|
if (ox2 > ox1) {
|
|
const float t0 = (cx1 - ox1) / (ox2 - ox1);
|
|
const float t1 = (cx2 - ox1) / (ox2 - ox1);
|
|
const float du = eu - su;
|
|
su = cmd.su + du * t0;
|
|
eu = cmd.su + du * t1;
|
|
const float mdu = meu - msu;
|
|
msu = cmd.mu[0] + mdu * t0;
|
|
meu = cmd.mu[0] + mdu * t1;
|
|
}
|
|
if (oy2 > oy1) {
|
|
const float t0 = (cy1 - oy1) / (oy2 - oy1);
|
|
const float t1 = (cy2 - oy1) / (oy2 - oy1);
|
|
const float dv = ev - sv;
|
|
sv = cmd.sv + dv * t0;
|
|
ev = cmd.sv + dv * t1;
|
|
const float mdv = mev - msv;
|
|
msv = cmd.mv[0] + mdv * t0;
|
|
mev = cmd.mv[0] + mdv * t1;
|
|
}
|
|
qx[0] = cx1; qy[0] = cy1;
|
|
qx[1] = cx2; qy[1] = cy1;
|
|
qx[2] = cx1; qy[2] = cy2;
|
|
qx[3] = cx2; qy[3] = cy2;
|
|
mu[0] = msu; mv[0] = msv;
|
|
mu[1] = meu; mv[1] = msv;
|
|
mu[2] = msu; mv[2] = mev;
|
|
mu[3] = meu; mv[3] = mev;
|
|
} else {
|
|
if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2))
|
|
continue;
|
|
for (int k = 0; k < 4; ++k) {
|
|
qx[k] = cmd.qx[k];
|
|
qy[k] = cmd.qy[k];
|
|
}
|
|
}
|
|
u[0] = su; v[0] = sv;
|
|
u[1] = eu; v[1] = sv;
|
|
u[2] = su; v[2] = ev;
|
|
u[3] = eu; v[3] = ev;
|
|
} else if (cmd.kind == UIRenderCommand::Bar && cmd.quad) {
|
|
// An untextured polygon piece (CPythonGraphic::RenderCoolTimeBox's fan triangles).
|
|
if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2))
|
|
continue;
|
|
for (int k = 0; k < 4; ++k) {
|
|
qx[k] = cmd.qx[k];
|
|
qy[k] = cmd.qy[k];
|
|
u[k] = 0.0f;
|
|
v[k] = 0.0f;
|
|
}
|
|
} else {
|
|
if (has_clip) {
|
|
x1 = std::max(x1, cmd.clip_x1);
|
|
y1 = std::max(y1, cmd.clip_y1);
|
|
x2 = std::min(x2, cmd.clip_x2);
|
|
y2 = std::min(y2, cmd.clip_y2);
|
|
}
|
|
if (x2 <= x1 || y2 <= y1) continue;
|
|
qx[0] = x1; qy[0] = y1;
|
|
qx[1] = x2; qy[1] = y1;
|
|
qx[2] = x1; qy[2] = y2;
|
|
qx[3] = x2; qy[3] = y2;
|
|
u[0] = su; v[0] = sv;
|
|
u[1] = eu; v[1] = sv;
|
|
u[2] = su; v[2] = ev;
|
|
u[3] = eu; v[3] = ev;
|
|
}
|
|
|
|
const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string();
|
|
const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string();
|
|
const bool is_masked = !mask_name.empty();
|
|
// CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use
|
|
// INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR.
|
|
const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image
|
|
? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1))
|
|
: 1;
|
|
const VkPipeline pipeline = get_pipeline(0, 0, blend_mode);
|
|
const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures);
|
|
|
|
for (int k = 0; k < 4; ++k) {
|
|
Vertex& vert = vertices[v_count + k];
|
|
vert.position[0] = qx[k] * inv_w - 1.0f;
|
|
vert.position[1] = 1.0f - qy[k] * inv_h;
|
|
vert.position[2] = 0.0f;
|
|
vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f;
|
|
vert.uv[0] = u[k]; vert.uv[1] = v[k];
|
|
const auto& col = (k < 2) ? top_color : bot_color;
|
|
vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3];
|
|
vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0;
|
|
vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f;
|
|
vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k];
|
|
vert.rhw = 1.0f;
|
|
}
|
|
indices[i_count + 0] = v_count + 0;
|
|
indices[i_count + 1] = v_count + 1;
|
|
indices[i_count + 2] = v_count + 2;
|
|
indices[i_count + 3] = v_count + 2;
|
|
indices[i_count + 4] = v_count + 1;
|
|
indices[i_count + 3 + 2] = v_count + 3;
|
|
|
|
const float mask_flag = is_masked ? 1.0f : 0.0f;
|
|
if (!batches.empty() &&
|
|
batches.back().behind_3d == cmd.behind_3d &&
|
|
batches.back().pipeline == pipeline &&
|
|
batches.back().descriptor == descriptor &&
|
|
batches.back().constants.params[2] == mask_flag) {
|
|
batches.back().index_count += 6;
|
|
} else {
|
|
UiBatch batch{};
|
|
batch.first_index = i_count;
|
|
batch.index_count = 6;
|
|
batch.pipeline = pipeline;
|
|
batch.descriptor = descriptor;
|
|
batch.behind_3d = cmd.behind_3d;
|
|
batch.constants.mvp = kIdentityMatrix;
|
|
batch.constants.params = {0.0f, 0.0f, mask_flag, 0.0f};
|
|
batches.push_back(batch);
|
|
}
|
|
v_count += 4;
|
|
i_count += 6;
|
|
++out_quad_count;
|
|
}
|
|
}
|
|
|
|
std::uint32_t find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const {
|
|
VkPhysicalDeviceMemoryProperties properties{};
|
|
vkGetPhysicalDeviceMemoryProperties(physical_, &properties);
|
|
for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i)
|
|
if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags)
|
|
return i;
|
|
throw std::runtime_error("no matching Vulkan memory type");
|
|
}
|
|
|
|
void select_device() {
|
|
std::uint32_t count = 0;
|
|
check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count");
|
|
if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK");
|
|
std::vector<VkPhysicalDevice> candidates(count);
|
|
check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices");
|
|
for (auto candidate : candidates) {
|
|
std::uint32_t family_count = 0;
|
|
vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr);
|
|
std::vector<VkQueueFamilyProperties> families(family_count);
|
|
vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data());
|
|
for (std::uint32_t i = 0; i < family_count; ++i) {
|
|
VkBool32 present = VK_FALSE;
|
|
check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support");
|
|
if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) {
|
|
physical_ = candidate; queue_family_ = i; break;
|
|
}
|
|
}
|
|
if (physical_) break;
|
|
}
|
|
if (!physical_) throw std::runtime_error("no graphics/present queue family");
|
|
VkPhysicalDeviceProperties properties{};
|
|
vkGetPhysicalDeviceProperties(physical_, &properties);
|
|
device_name_ = properties.deviceName;
|
|
timestamp_period_ns_ = properties.limits.timestampPeriod;
|
|
VkPhysicalDeviceFeatures supported_features{};
|
|
vkGetPhysicalDeviceFeatures(physical_, &supported_features);
|
|
VkPhysicalDeviceFeatures enabled_features{};
|
|
if (supported_features.samplerAnisotropy) {
|
|
enabled_features.samplerAnisotropy = VK_TRUE;
|
|
max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy);
|
|
}
|
|
// Multisampled 3D edges; MT_MSAA=1 keeps the single-sample path, 2/4/8 request a count.
|
|
{
|
|
int wanted = 4;
|
|
if (const char* env = std::getenv("MT_MSAA")) wanted = std::max(1, std::atoi(env));
|
|
const VkSampleCountFlags supported =
|
|
properties.limits.framebufferColorSampleCounts & properties.limits.framebufferDepthSampleCounts;
|
|
samples_ = VK_SAMPLE_COUNT_1_BIT;
|
|
for (VkSampleCountFlagBits c : {VK_SAMPLE_COUNT_8_BIT, VK_SAMPLE_COUNT_4_BIT, VK_SAMPLE_COUNT_2_BIT})
|
|
if (int(c) <= wanted && (supported & c)) { samples_ = c; break; }
|
|
}
|
|
std::uint32_t extension_count = 0;
|
|
check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count");
|
|
std::vector<VkExtensionProperties> available(extension_count);
|
|
check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions");
|
|
std::vector<const char*> extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME};
|
|
constexpr const char* portability_subset = "VK_KHR_portability_subset";
|
|
for (const auto& extension : available)
|
|
if (std::strcmp(extension.extensionName, portability_subset) == 0)
|
|
extensions.push_back(portability_subset);
|
|
const float priority = 1.0f;
|
|
VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO};
|
|
queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority;
|
|
VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO};
|
|
device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info;
|
|
device_info.enabledExtensionCount = static_cast<std::uint32_t>(extensions.size());
|
|
device_info.ppEnabledExtensionNames = extensions.data();
|
|
device_info.pEnabledFeatures = &enabled_features;
|
|
check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice");
|
|
vkGetDeviceQueue(device_, queue_family_, 0, &queue_);
|
|
}
|
|
|
|
void create_swapchain() {
|
|
VkSurfaceCapabilitiesKHR capabilities{};
|
|
check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities");
|
|
std::uint32_t count = 0;
|
|
check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count");
|
|
if (!count) throw std::runtime_error("surface has no formats");
|
|
std::vector<VkSurfaceFormatKHR> formats(count);
|
|
check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats");
|
|
VkSurfaceFormatKHR format = formats.front();
|
|
for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate;
|
|
swapchain_format_ = format.format;
|
|
int width = 0, height = 0;
|
|
SDL_GetWindowSizeInPixels(window_, &width, &height);
|
|
if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent;
|
|
else {
|
|
extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width);
|
|
extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height);
|
|
}
|
|
|
|
present_mode_ = VK_PRESENT_MODE_FIFO_KHR;
|
|
if (!vsync_) {
|
|
std::uint32_t mode_count = 0;
|
|
check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count");
|
|
std::vector<VkPresentModeKHR> modes(mode_count);
|
|
if (mode_count)
|
|
check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes");
|
|
if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end())
|
|
present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR;
|
|
else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end())
|
|
present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR;
|
|
}
|
|
|
|
VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR};
|
|
info.surface = surface_;
|
|
info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX);
|
|
info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_;
|
|
info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT |
|
|
(capabilities.supportedUsageFlags & VK_IMAGE_USAGE_TRANSFER_SRC_BIT);
|
|
info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
// Android surfaces can report a quarter-turn transform even when SDL's window is
|
|
// already landscape. Rendering in SDL window coordinates and requesting that transform
|
|
// again rotates the entire game image on presentation.
|
|
info.preTransform = (capabilities.supportedTransforms & VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR)
|
|
? VK_SURFACE_TRANSFORM_IDENTITY_BIT_KHR : capabilities.currentTransform;
|
|
#ifdef __ANDROID__
|
|
SDL_Log("swapchain: window=%dx%d extent=%ux%u currentTransform=%u preTransform=%u",
|
|
width, height, extent_.width, extent_.height,
|
|
unsigned(capabilities.currentTransform), unsigned(info.preTransform));
|
|
#endif
|
|
info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_;
|
|
info.clipped = VK_TRUE;
|
|
check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR");
|
|
check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count");
|
|
std::vector<VkImage> images(count);
|
|
check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images");
|
|
swapchain_images_ = images;
|
|
for (auto image : images) {
|
|
VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
|
|
view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_;
|
|
view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
|
view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1;
|
|
VkImageView handle = VK_NULL_HANDLE;
|
|
check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView");
|
|
image_views_.push_back(handle);
|
|
}
|
|
VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
|
|
depth_info.imageType = VK_IMAGE_TYPE_2D;
|
|
depth_info.format = VK_FORMAT_D32_SFLOAT;
|
|
depth_info.extent = {extent_.width, extent_.height, 1};
|
|
depth_info.mipLevels = 1; depth_info.arrayLayers = 1;
|
|
depth_info.samples = samples_;
|
|
depth_info.tiling = VK_IMAGE_TILING_OPTIMAL;
|
|
depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
|
|
depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth");
|
|
VkMemoryRequirements depth_reqs{};
|
|
vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs);
|
|
VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
depth_alloc.allocationSize = depth_reqs.size;
|
|
depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
|
check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth");
|
|
check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth");
|
|
VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
|
|
depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
|
depth_view.format = VK_FORMAT_D32_SFLOAT;
|
|
depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT;
|
|
depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1;
|
|
check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth");
|
|
if (samples_ != VK_SAMPLE_COUNT_1_BIT) create_msaa_color();
|
|
}
|
|
|
|
// The multisampled colour target the subpass resolves into the swapchain image. It never
|
|
// leaves the render pass, so it is transient (lazily allocated tile memory where available).
|
|
void create_msaa_color() {
|
|
VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
|
|
info.imageType = VK_IMAGE_TYPE_2D;
|
|
info.format = swapchain_format_;
|
|
info.extent = {extent_.width, extent_.height, 1};
|
|
info.mipLevels = 1; info.arrayLayers = 1;
|
|
info.samples = samples_;
|
|
info.tiling = VK_IMAGE_TILING_OPTIMAL;
|
|
info.usage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSIENT_ATTACHMENT_BIT;
|
|
info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
check(vkCreateImage(device_, &info, nullptr, &msaa_image_), "vkCreateImage msaa color");
|
|
VkMemoryRequirements reqs{};
|
|
vkGetImageMemoryRequirements(device_, msaa_image_, &reqs);
|
|
VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
alloc.allocationSize = reqs.size;
|
|
try {
|
|
alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_LAZILY_ALLOCATED_BIT);
|
|
} catch (const std::runtime_error&) {
|
|
alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
|
}
|
|
check(vkAllocateMemory(device_, &alloc, nullptr, &msaa_memory_), "vkAllocateMemory msaa color");
|
|
check(vkBindImageMemory(device_, msaa_image_, msaa_memory_, 0), "vkBindImageMemory msaa color");
|
|
VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
|
|
view.image = msaa_image_; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_;
|
|
view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
|
view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1;
|
|
check(vkCreateImageView(device_, &view, nullptr, &msaa_view_), "vkCreateImageView msaa color");
|
|
}
|
|
|
|
void destroy_msaa_color() {
|
|
if (msaa_view_) { vkDestroyImageView(device_, msaa_view_, nullptr); msaa_view_ = VK_NULL_HANDLE; }
|
|
if (msaa_image_) { vkDestroyImage(device_, msaa_image_, nullptr); msaa_image_ = VK_NULL_HANDLE; }
|
|
if (msaa_memory_) { vkFreeMemory(device_, msaa_memory_, nullptr); msaa_memory_ = VK_NULL_HANDLE; }
|
|
}
|
|
|
|
void create_framebuffers() {
|
|
for (auto view : image_views_) {
|
|
// Single-sample: {swapchain, depth}. MSAA: {msaa colour, depth, swapchain resolve}.
|
|
VkImageView fb_attachments[3] = {view, depth_view_, VK_NULL_HANDLE};
|
|
std::uint32_t count = 2;
|
|
if (samples_ != VK_SAMPLE_COUNT_1_BIT) {
|
|
fb_attachments[0] = msaa_view_;
|
|
fb_attachments[2] = view;
|
|
count = 3;
|
|
}
|
|
VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO};
|
|
framebuffer.renderPass = pass_; framebuffer.attachmentCount = count; framebuffer.pAttachments = fb_attachments;
|
|
framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1;
|
|
VkFramebuffer handle = VK_NULL_HANDLE;
|
|
check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer");
|
|
framebuffers_.push_back(handle);
|
|
}
|
|
}
|
|
|
|
void create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) {
|
|
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
|
|
info.size = size;
|
|
info.usage = usage;
|
|
info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host");
|
|
VkMemoryRequirements reqs{};
|
|
vkGetBufferMemoryRequirements(device_, buffer, &reqs);
|
|
VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
alloc.allocationSize = reqs.size;
|
|
alloc.memoryTypeIndex = find_memory_type(
|
|
reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
|
|
check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host");
|
|
check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host");
|
|
check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host");
|
|
}
|
|
|
|
void create_descriptors_and_buffers() {
|
|
VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO};
|
|
sampler_info.magFilter = VK_FILTER_LINEAR;
|
|
sampler_info.minFilter = VK_FILTER_LINEAR;
|
|
sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR;
|
|
sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
|
sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
|
sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
|
sampler_info.minLod = 0.0f;
|
|
sampler_info.maxLod = VK_LOD_CLAMP_NONE;
|
|
if (max_anisotropy_ > 1.0f) {
|
|
sampler_info.anisotropyEnable = VK_TRUE;
|
|
sampler_info.maxAnisotropy = max_anisotropy_;
|
|
}
|
|
check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler");
|
|
|
|
VkSamplerCreateInfo clamp_info = sampler_info;
|
|
clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp");
|
|
|
|
VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO};
|
|
ui_sampler_info.magFilter = VK_FILTER_LINEAR;
|
|
ui_sampler_info.minFilter = VK_FILTER_LINEAR;
|
|
ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST;
|
|
ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
ui_sampler_info.minLod = 0.0f;
|
|
ui_sampler_info.maxLod = 0.0f;
|
|
ui_sampler_info.anisotropyEnable = VK_FALSE;
|
|
ui_sampler_info.maxAnisotropy = 1.0f;
|
|
check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui");
|
|
|
|
VkDescriptorSetLayoutBinding tex_bindings[2]{};
|
|
tex_bindings[0].binding = 0;
|
|
tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
|
tex_bindings[0].descriptorCount = 1;
|
|
tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT;
|
|
tex_bindings[1].binding = 1;
|
|
tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
|
tex_bindings[1].descriptorCount = 1;
|
|
tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT;
|
|
VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO};
|
|
layout_info.bindingCount = 2;
|
|
layout_info.pBindings = tex_bindings;
|
|
check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex");
|
|
|
|
VkDescriptorSetLayoutBinding bone_bindings[2]{};
|
|
bone_bindings[0].binding = 0;
|
|
bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
|
|
bone_bindings[0].descriptorCount = 1;
|
|
bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT;
|
|
bone_bindings[1].binding = 1;
|
|
bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC;
|
|
bone_bindings[1].descriptorCount = 1;
|
|
bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT;
|
|
VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO};
|
|
bone_layout_info.bindingCount = 2;
|
|
bone_layout_info.pBindings = bone_bindings;
|
|
check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone");
|
|
|
|
VkDescriptorPoolSize pool_sizes[3] = {
|
|
{VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384},
|
|
{VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4 * kMaxFramesInFlight},
|
|
{VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4 * kMaxFramesInFlight}};
|
|
VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO};
|
|
pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT;
|
|
pool_info.maxSets = 8192;
|
|
pool_info.poolSizeCount = 3;
|
|
pool_info.pPoolSizes = pool_sizes;
|
|
check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool");
|
|
|
|
const std::uint8_t white_pixel[4] = {255, 255, 255, 255};
|
|
upload_texture(1, 1, white_pixel, fallback_texture_, false, false);
|
|
fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view);
|
|
fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view);
|
|
|
|
VkPhysicalDeviceProperties properties{};
|
|
vkGetPhysicalDeviceProperties(physical_, &properties);
|
|
const auto alignment = std::max<VkDeviceSize>(
|
|
1, properties.limits.minStorageBufferOffsetAlignment);
|
|
state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment;
|
|
staging_alignment_ = std::max<VkDeviceSize>(16, properties.limits.optimalBufferCopyOffsetAlignment);
|
|
|
|
for (auto& slot : frames_) {
|
|
void* bone_raw = nullptr;
|
|
create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, slot.bone_buffer, slot.bone_memory, &bone_raw);
|
|
slot.bone_mapped = static_cast<float*>(bone_raw);
|
|
slot.bone_capacity = kMaxBonesPerFrame;
|
|
for (std::size_t i = 0; i < 16; ++i) slot.bone_mapped[i] = kIdentityMatrix[i];
|
|
|
|
slot.state_capacity = 256;
|
|
void* state_raw = nullptr;
|
|
create_host_buffer(slot.state_capacity * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
|
|
slot.state_buffer, slot.state_memory, &state_raw);
|
|
slot.state_mapped = static_cast<std::uint8_t*>(state_raw);
|
|
|
|
VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO};
|
|
bone_alloc.descriptorPool = descriptor_pool_;
|
|
bone_alloc.descriptorSetCount = 1;
|
|
bone_alloc.pSetLayouts = &bone_descriptor_layout_;
|
|
check(vkAllocateDescriptorSets(device_, &bone_alloc, &slot.bone_descriptor_set), "vkAllocateDescriptorSets bone");
|
|
|
|
VkDescriptorBufferInfo buffer_infos[2] = {
|
|
{slot.bone_buffer, 0, kBoneBufferBytes},
|
|
{slot.state_buffer, 0, sizeof(FixedFunctionState)}};
|
|
VkWriteDescriptorSet writes[2]{};
|
|
for (int binding = 0; binding < 2; ++binding) {
|
|
writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
|
|
writes[binding].dstSet = slot.bone_descriptor_set;
|
|
writes[binding].dstBinding = static_cast<std::uint32_t>(binding);
|
|
writes[binding].descriptorCount = 1;
|
|
writes[binding].descriptorType = binding == 0
|
|
? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC;
|
|
writes[binding].pBufferInfo = &buffer_infos[binding];
|
|
}
|
|
vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr);
|
|
|
|
void* ui_raw = nullptr;
|
|
create_host_buffer(
|
|
kUiBufferBytes,
|
|
VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT,
|
|
slot.ui_buffer,
|
|
slot.ui_memory,
|
|
&ui_raw);
|
|
slot.ui_mapped = static_cast<std::uint8_t*>(ui_raw);
|
|
slot.ui_vertex_capacity = kMaxUiVerticesPerFrame;
|
|
slot.ui_index_capacity = kMaxUiIndicesPerFrame;
|
|
slot.ui_vertex_bytes = kUiVertexBytes;
|
|
slot.staging_wanted = kStagingRingBytes;
|
|
}
|
|
}
|
|
|
|
// The slot's fence has signalled: its oversize staging buffers are free, and the ring restarts
|
|
// (grown to the largest frame's uploads seen so far, so texture streaming stays in the ring).
|
|
void release_retired_buffers(FrameSlot& slot) {
|
|
for (auto [buffer, memory] : slot.retired_buffers) {
|
|
vkDestroyBuffer(device_, buffer, nullptr);
|
|
vkFreeMemory(device_, memory, nullptr);
|
|
}
|
|
slot.retired_buffers.clear();
|
|
}
|
|
|
|
void reset_staging(FrameSlot& slot) {
|
|
release_retired_buffers(slot);
|
|
slot.staging_used = 0;
|
|
if (slot.staging_wanted <= slot.staging_capacity) return;
|
|
if (slot.staging_mapped) vkUnmapMemory(device_, slot.staging_memory);
|
|
if (slot.staging_buffer) vkDestroyBuffer(device_, slot.staging_buffer, nullptr);
|
|
if (slot.staging_memory) vkFreeMemory(device_, slot.staging_memory, nullptr);
|
|
void* raw = nullptr;
|
|
slot.staging_capacity = std::min(slot.staging_wanted, kStagingRingMaxBytes);
|
|
create_host_buffer(slot.staging_capacity, VK_BUFFER_USAGE_TRANSFER_SRC_BIT,
|
|
slot.staging_buffer, slot.staging_memory, &raw);
|
|
slot.staging_mapped = static_cast<std::uint8_t*>(raw);
|
|
slot.staging_wanted = slot.staging_capacity;
|
|
}
|
|
|
|
void ensure_bone_capacity(std::size_t required) {
|
|
if (required <= f_->bone_capacity) return;
|
|
VkPhysicalDeviceProperties properties{};
|
|
vkGetPhysicalDeviceProperties(physical_, &properties);
|
|
const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float));
|
|
if (required > max_bones || required > UINT32_MAX)
|
|
throw std::runtime_error("bone palette exceeds device storage-buffer limit");
|
|
std::size_t next = std::min(max_bones, std::max(required, f_->bone_capacity * 2));
|
|
VkBuffer buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory memory = VK_NULL_HANDLE;
|
|
void* mapped = nullptr;
|
|
create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped);
|
|
vkUnmapMemory(device_, f_->bone_memory);
|
|
vkDestroyBuffer(device_, f_->bone_buffer, nullptr);
|
|
vkFreeMemory(device_, f_->bone_memory, nullptr);
|
|
f_->bone_buffer = buffer;
|
|
f_->bone_memory = memory;
|
|
f_->bone_mapped = static_cast<float*>(mapped);
|
|
f_->bone_capacity = next;
|
|
VkDescriptorBufferInfo info{f_->bone_buffer, 0, next * 16 * sizeof(float)};
|
|
VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET};
|
|
write.dstSet = f_->bone_descriptor_set;
|
|
write.dstBinding = 0;
|
|
write.descriptorCount = 1;
|
|
write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
|
|
write.pBufferInfo = &info;
|
|
vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr);
|
|
}
|
|
|
|
void ensure_state_capacity(std::size_t required) {
|
|
if (required <= f_->state_capacity) return;
|
|
const auto next = std::max(required, f_->state_capacity * 2);
|
|
if (next > UINT32_MAX / state_stride_)
|
|
throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range");
|
|
VkBuffer buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory memory = VK_NULL_HANDLE;
|
|
void* mapped = nullptr;
|
|
create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
|
|
buffer, memory, &mapped);
|
|
vkUnmapMemory(device_, f_->state_memory);
|
|
vkDestroyBuffer(device_, f_->state_buffer, nullptr);
|
|
vkFreeMemory(device_, f_->state_memory, nullptr);
|
|
f_->state_buffer = buffer;
|
|
f_->state_memory = memory;
|
|
f_->state_mapped = static_cast<std::uint8_t*>(mapped);
|
|
f_->state_capacity = next;
|
|
VkDescriptorBufferInfo info{f_->state_buffer, 0, sizeof(FixedFunctionState)};
|
|
VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET};
|
|
write.dstSet = f_->bone_descriptor_set;
|
|
write.dstBinding = 1;
|
|
write.descriptorCount = 1;
|
|
write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC;
|
|
write.pBufferInfo = &info;
|
|
vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr);
|
|
}
|
|
|
|
void ensure_ui_capacity(std::size_t command_count) {
|
|
if (command_count <= f_->ui_vertex_capacity / 4 && command_count <= f_->ui_index_capacity / 6) return;
|
|
if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit");
|
|
const std::size_t quads = std::max(command_count, f_->ui_vertex_capacity / 2);
|
|
const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex);
|
|
const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t);
|
|
if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull)
|
|
throw std::runtime_error("UI geometry exceeds 128 MiB safety limit");
|
|
VkBuffer buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory memory = VK_NULL_HANDLE;
|
|
void* mapped = nullptr;
|
|
create_host_buffer(vertex_bytes + index_bytes,
|
|
VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped);
|
|
vkUnmapMemory(device_, f_->ui_memory);
|
|
vkDestroyBuffer(device_, f_->ui_buffer, nullptr);
|
|
vkFreeMemory(device_, f_->ui_memory, nullptr);
|
|
f_->ui_buffer = buffer;
|
|
f_->ui_memory = memory;
|
|
f_->ui_mapped = static_cast<std::uint8_t*>(mapped);
|
|
f_->ui_vertex_capacity = quads * 4;
|
|
f_->ui_index_capacity = quads * 6;
|
|
f_->ui_vertex_bytes = vertex_bytes;
|
|
}
|
|
|
|
void create_render_pass_and_layout() {
|
|
const bool msaa = samples_ != VK_SAMPLE_COUNT_1_BIT;
|
|
VkAttachmentDescription attachments[3]{};
|
|
attachments[0].format = swapchain_format_; attachments[0].samples = samples_;
|
|
attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR;
|
|
attachments[0].storeOp = msaa ? VK_ATTACHMENT_STORE_OP_DONT_CARE : VK_ATTACHMENT_STORE_OP_STORE;
|
|
attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
|
attachments[0].finalLayout = msaa ? VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL : VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
|
|
// MSAA resolve target: the swapchain image.
|
|
attachments[2].format = swapchain_format_; attachments[2].samples = VK_SAMPLE_COUNT_1_BIT;
|
|
attachments[2].loadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE; attachments[2].storeOp = VK_ATTACHMENT_STORE_OP_STORE;
|
|
attachments[2].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[2].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR;
|
|
attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = samples_;
|
|
attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
|
|
attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
|
attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
|
|
|
VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL};
|
|
VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL};
|
|
VkAttachmentReference resolve_ref{2, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL};
|
|
VkSubpassDescription subpass{};
|
|
subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
|
|
subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref;
|
|
if (msaa) subpass.pResolveAttachments = &resolve_ref;
|
|
subpass.pDepthStencilAttachment = &depth_ref;
|
|
|
|
VkSubpassDependency dependency{};
|
|
dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0;
|
|
dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT;
|
|
dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT;
|
|
dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
|
|
|
|
VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO};
|
|
pass_info.attachmentCount = msaa ? 3 : 2; pass_info.pAttachments = attachments;
|
|
pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass;
|
|
pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency;
|
|
check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass");
|
|
create_framebuffers();
|
|
|
|
const auto vert = read_spirv("native.vert.spv");
|
|
const auto frag = read_spirv("native.frag.spv");
|
|
auto module = [&](const std::vector<std::uint32_t>& code) {
|
|
VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO};
|
|
info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data();
|
|
VkShaderModule handle = VK_NULL_HANDLE;
|
|
check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule");
|
|
return handle;
|
|
};
|
|
vertex_module_ = module(vert);
|
|
fragment_module_ = module(frag);
|
|
|
|
VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_};
|
|
VkPushConstantRange push_constants{};
|
|
push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT;
|
|
push_constants.size = sizeof(PushConstants);
|
|
VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO};
|
|
layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts;
|
|
layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants;
|
|
check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout");
|
|
|
|
get_pipeline(0, 2, 0);
|
|
create_offscreen_pass();
|
|
}
|
|
|
|
// The render pass of offscreen render-target textures: colour (R5G6B5 like the D3D surface when
|
|
// the device can render to it) + D32 depth, both loaded and stored, colour left shader-readable.
|
|
void create_offscreen_pass() {
|
|
VkFormatProperties properties{};
|
|
vkGetPhysicalDeviceFormatProperties(physical_, VK_FORMAT_R5G6B5_UNORM_PACK16, &properties);
|
|
const VkFormatFeatureFlags wanted =
|
|
VK_FORMAT_FEATURE_COLOR_ATTACHMENT_BIT | VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT;
|
|
offscreen_format_ = (properties.optimalTilingFeatures & wanted) == wanted
|
|
? VK_FORMAT_R5G6B5_UNORM_PACK16 : VK_FORMAT_R8G8B8A8_UNORM;
|
|
VkAttachmentDescription attachments[2]{};
|
|
attachments[0].format = offscreen_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT;
|
|
attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE;
|
|
attachments[0].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
|
|
attachments[0].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
|
|
attachments[0].initialLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
attachments[0].finalLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT;
|
|
attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_LOAD; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_STORE;
|
|
attachments[1].stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
|
|
attachments[1].stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
|
|
attachments[1].initialLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
|
attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
|
VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL};
|
|
VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL};
|
|
VkSubpassDescription subpass{};
|
|
subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
|
|
subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref;
|
|
subpass.pDepthStencilAttachment = &depth_ref;
|
|
VkSubpassDependency dependencies[2]{};
|
|
// Earlier work on the queue -- the previous frame sampling the texture, the previous
|
|
// offscreen pass, the first-use clear -- completes before this pass writes it.
|
|
dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL; dependencies[0].dstSubpass = 0;
|
|
dependencies[0].srcStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
|
|
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT |
|
|
VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT;
|
|
dependencies[0].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
|
|
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
dependencies[0].dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT |
|
|
VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT;
|
|
dependencies[0].dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
|
|
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
|
|
// The back-buffer pass samples the result.
|
|
dependencies[1].srcSubpass = 0; dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL;
|
|
dependencies[1].srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT;
|
|
dependencies[1].srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
|
|
dependencies[1].dstStageMask = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
|
|
dependencies[1].dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
|
|
VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO};
|
|
pass_info.attachmentCount = 2; pass_info.pAttachments = attachments;
|
|
pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass;
|
|
pass_info.dependencyCount = 2; pass_info.pDependencies = dependencies;
|
|
check(vkCreateRenderPass(device_, &pass_info, nullptr, &offscreen_pass_), "vkCreateRenderPass offscreen");
|
|
}
|
|
|
|
// "rt:<id>:<w>x<h>" -> its render target, created on first use (by a draw into it or by a draw
|
|
// sampling it); null for a malformed name.
|
|
RenderTarget* get_render_target(const std::string& name) {
|
|
if (auto it = render_targets_.find(name); it != render_targets_.end()) return &it->second;
|
|
unsigned id = 0, w = 0, h = 0;
|
|
if (std::sscanf(name.c_str(), "rt:%u:%ux%u", &id, &w, &h) != 3 || !w || !h || w > 4096 || h > 4096)
|
|
return nullptr;
|
|
RenderTarget& target = render_targets_[name];
|
|
target.width = w;
|
|
target.height = h;
|
|
auto make_image = [&](VkFormat format, VkImageUsageFlags usage, VkImageAspectFlags aspect,
|
|
VkImage& image, VkDeviceMemory& memory, VkImageView& view) {
|
|
VkImageCreateInfo info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
|
|
info.imageType = VK_IMAGE_TYPE_2D;
|
|
info.format = format;
|
|
info.extent = {w, h, 1};
|
|
info.mipLevels = 1; info.arrayLayers = 1;
|
|
info.samples = VK_SAMPLE_COUNT_1_BIT;
|
|
info.tiling = VK_IMAGE_TILING_OPTIMAL;
|
|
info.usage = usage;
|
|
info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
check(vkCreateImage(device_, &info, nullptr, &image), "vkCreateImage render target");
|
|
VkMemoryRequirements reqs{};
|
|
vkGetImageMemoryRequirements(device_, image, &reqs);
|
|
VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
alloc.allocationSize = reqs.size;
|
|
alloc.memoryTypeIndex = find_memory_type(reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
|
check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory render target");
|
|
check(vkBindImageMemory(device_, image, memory, 0), "vkBindImageMemory render target");
|
|
VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
|
|
view_info.image = image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; view_info.format = format;
|
|
view_info.subresourceRange = {aspect, 0, 1, 0, 1};
|
|
check(vkCreateImageView(device_, &view_info, nullptr, &view), "vkCreateImageView render target");
|
|
};
|
|
make_image(offscreen_format_,
|
|
VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT |
|
|
VK_IMAGE_USAGE_TRANSFER_SRC_BIT,
|
|
VK_IMAGE_ASPECT_COLOR_BIT, target.texture.image, target.texture.memory, target.texture.view);
|
|
make_image(VK_FORMAT_D32_SFLOAT,
|
|
VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT,
|
|
VK_IMAGE_ASPECT_DEPTH_BIT, target.depth_image, target.depth_memory, target.depth_view);
|
|
const VkImageView views[2] = {target.texture.view, target.depth_view};
|
|
VkFramebufferCreateInfo fb{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO};
|
|
fb.renderPass = offscreen_pass_;
|
|
fb.attachmentCount = 2; fb.pAttachments = views;
|
|
fb.width = w; fb.height = h; fb.layers = 1;
|
|
check(vkCreateFramebuffer(device_, &fb, nullptr, &target.framebuffer), "vkCreateFramebuffer render target");
|
|
target.texture.descriptor = allocate_texture_descriptor_set(target.texture.view, fallback_texture_.view);
|
|
target.texture.ui_descriptor = allocate_ui_texture_descriptor_set(target.texture.view, fallback_texture_.view);
|
|
return ⌖
|
|
}
|
|
|
|
void release_render_target(RenderTarget& target) {
|
|
if (target.framebuffer) vkDestroyFramebuffer(device_, target.framebuffer, nullptr);
|
|
if (target.depth_view) vkDestroyImageView(device_, target.depth_view, nullptr);
|
|
if (target.depth_image) vkDestroyImage(device_, target.depth_image, nullptr);
|
|
if (target.depth_memory) vkFreeMemory(device_, target.depth_memory, nullptr);
|
|
release_texture(target.texture);
|
|
target = {};
|
|
}
|
|
|
|
// A new render target starts white with depth 1 (what the CPU surface's first Clear leaves), in
|
|
// the layouts offscreen_pass_ expects. Recorded outside any render pass.
|
|
void initialize_render_target(RenderTarget& target) {
|
|
VkImageMemoryBarrier barriers[2]{};
|
|
for (auto& b : barriers) {
|
|
b.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER;
|
|
b.srcQueueFamilyIndex = b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
b.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
|
b.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
|
b.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
}
|
|
barriers[0].image = target.texture.image;
|
|
barriers[0].subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1};
|
|
barriers[1].image = target.depth_image;
|
|
barriers[1].subresourceRange = {VK_IMAGE_ASPECT_DEPTH_BIT, 0, 1, 0, 1};
|
|
vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT,
|
|
0, 0, nullptr, 0, nullptr, 2, barriers);
|
|
const VkClearColorValue white{{1.0f, 1.0f, 1.0f, 1.0f}};
|
|
const VkClearDepthStencilValue far{1.0f, 0};
|
|
vkCmdClearColorImage(f_->command, target.texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &white, 1,
|
|
&barriers[0].subresourceRange);
|
|
vkCmdClearDepthStencilImage(f_->command, target.depth_image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, &far, 1,
|
|
&barriers[1].subresourceRange);
|
|
for (auto& b : barriers) {
|
|
b.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
|
b.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
}
|
|
barriers[0].newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
barriers[0].dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_READ_BIT;
|
|
barriers[1].newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL;
|
|
barriers[1].dstAccessMask = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT;
|
|
vkCmdPipelineBarrier(f_->command, VK_PIPELINE_STAGE_TRANSFER_BIT,
|
|
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT |
|
|
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT,
|
|
0, 0, nullptr, 0, nullptr, 2, barriers);
|
|
target.initialized = true;
|
|
}
|
|
|
|
// offscreen: for offscreen_pass_ (single-sampled render-target texture) instead of pass_.
|
|
VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend,
|
|
bool lines = false, std::uint8_t z_func = 4, bool offscreen = false) {
|
|
const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) |
|
|
(std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) |
|
|
(std::uint32_t(z_func & 0xfu) << 25) | (offscreen ? (1u << 29) : 0u);
|
|
if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second;
|
|
|
|
VkPipelineShaderStageCreateInfo stages[2]{};
|
|
stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
|
|
stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main";
|
|
stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main";
|
|
|
|
VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX};
|
|
VkVertexInputAttributeDescription attributes[9] = {
|
|
{0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)},
|
|
{1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)},
|
|
{2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)},
|
|
{3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)},
|
|
{4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)},
|
|
{5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)},
|
|
{6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)},
|
|
{7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)},
|
|
{8, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, vertex_fog)}};
|
|
VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO};
|
|
input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding;
|
|
input.vertexAttributeDescriptionCount = 9; input.pVertexAttributeDescriptions = attributes;
|
|
|
|
VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO};
|
|
assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST;
|
|
|
|
VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1};
|
|
// An offscreen target is at most the device's 4096 (GetDeviceCaps); the viewport clips inside it.
|
|
VkRect2D scissor{{0, 0}, offscreen ? VkExtent2D{4096, 4096} : extent_};
|
|
VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO};
|
|
viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport;
|
|
viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor;
|
|
// D3DVIEWPORT8 changes per draw (SetViewport), so the viewport is dynamic state.
|
|
const VkDynamicState dynamic_states[] = {VK_DYNAMIC_STATE_VIEWPORT};
|
|
VkPipelineDynamicStateCreateInfo dynamic_info{VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO};
|
|
dynamic_info.dynamicStateCount = 1; dynamic_info.pDynamicStates = dynamic_states;
|
|
|
|
VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO};
|
|
raster.polygonMode = VK_POLYGON_MODE_FILL;
|
|
raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE);
|
|
raster.frontFace = VK_FRONT_FACE_CLOCKWISE;
|
|
raster.lineWidth = 1;
|
|
|
|
VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO};
|
|
multisample.rasterizationSamples = offscreen ? VK_SAMPLE_COUNT_1_BIT : samples_;
|
|
|
|
VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO};
|
|
depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE;
|
|
depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE;
|
|
switch (z_func) {
|
|
case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break;
|
|
case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break;
|
|
case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break;
|
|
case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break;
|
|
case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break;
|
|
case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break;
|
|
case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break;
|
|
default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break;
|
|
}
|
|
|
|
auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor {
|
|
switch (d3d) {
|
|
case 1: return VK_BLEND_FACTOR_ZERO;
|
|
case 2: return VK_BLEND_FACTOR_ONE;
|
|
case 3: return VK_BLEND_FACTOR_SRC_COLOR;
|
|
case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR;
|
|
case 5: return VK_BLEND_FACTOR_SRC_ALPHA;
|
|
case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA;
|
|
case 7: return VK_BLEND_FACTOR_DST_ALPHA;
|
|
case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA;
|
|
case 9: return VK_BLEND_FACTOR_DST_COLOR;
|
|
case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR;
|
|
case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE;
|
|
default: return fallback;
|
|
}
|
|
};
|
|
|
|
VkPipelineColorBlendAttachmentState blend_attachment{};
|
|
blend_attachment.colorWriteMask = 0xf;
|
|
if (blend > 0) {
|
|
blend_attachment.blendEnable = VK_TRUE;
|
|
if (blend >= 16) {
|
|
blend_attachment.srcColorBlendFactor =
|
|
d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA);
|
|
blend_attachment.dstColorBlendFactor =
|
|
d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA);
|
|
} else {
|
|
blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA;
|
|
blend_attachment.dstColorBlendFactor =
|
|
blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA;
|
|
}
|
|
blend_attachment.colorBlendOp = VK_BLEND_OP_ADD;
|
|
// D3D8 has no separate alpha blend: SRCBLEND/DESTBLEND apply to alpha as well.
|
|
auto alpha_factor = [](VkBlendFactor factor) {
|
|
switch (factor) {
|
|
case VK_BLEND_FACTOR_SRC_COLOR: return VK_BLEND_FACTOR_SRC_ALPHA;
|
|
case VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA;
|
|
case VK_BLEND_FACTOR_DST_COLOR: return VK_BLEND_FACTOR_DST_ALPHA;
|
|
case VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA;
|
|
case VK_BLEND_FACTOR_SRC_ALPHA_SATURATE: return VK_BLEND_FACTOR_ONE;
|
|
default: return factor;
|
|
}
|
|
};
|
|
blend_attachment.srcAlphaBlendFactor = alpha_factor(blend_attachment.srcColorBlendFactor);
|
|
blend_attachment.dstAlphaBlendFactor = alpha_factor(blend_attachment.dstColorBlendFactor);
|
|
blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD;
|
|
}
|
|
VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO};
|
|
blending.attachmentCount = 1; blending.pAttachments = &blend_attachment;
|
|
|
|
VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO};
|
|
pipeline_info.stageCount = 2; pipeline_info.pStages = stages;
|
|
pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly;
|
|
pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster;
|
|
pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil;
|
|
pipeline_info.pColorBlendState = &blending;
|
|
pipeline_info.pDynamicState = &dynamic_info;
|
|
pipeline_info.layout = layout_; pipeline_info.renderPass = offscreen ? offscreen_pass_ : pass_;
|
|
VkPipeline handle = VK_NULL_HANDLE;
|
|
check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines");
|
|
pipelines_.emplace(key, handle);
|
|
return handle;
|
|
}
|
|
|
|
VkDescriptorSet allocate_texture_descriptor_set(VkImageView view0, VkImageView view1,
|
|
VkSampler sampler0 = VK_NULL_HANDLE,
|
|
VkSampler sampler1 = VK_NULL_HANDLE) {
|
|
VkDescriptorSet set = VK_NULL_HANDLE;
|
|
VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO};
|
|
desc_alloc.descriptorPool = descriptor_pool_;
|
|
desc_alloc.descriptorSetCount = 1;
|
|
desc_alloc.pSetLayouts = &descriptor_layout_;
|
|
check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets");
|
|
|
|
VkDescriptorImageInfo image_infos[2]{};
|
|
image_infos[0].sampler = sampler0 ? sampler0 : sampler_;
|
|
image_infos[0].imageView = view0;
|
|
image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_);
|
|
image_infos[1].imageView = view1;
|
|
image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
|
|
VkWriteDescriptorSet writes[2]{};
|
|
for (int b = 0; b < 2; ++b) {
|
|
writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
|
|
writes[b].dstSet = set;
|
|
writes[b].dstBinding = static_cast<std::uint32_t>(b);
|
|
writes[b].descriptorCount = 1;
|
|
writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
|
writes[b].pImageInfo = &image_infos[b];
|
|
}
|
|
vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr);
|
|
return set;
|
|
}
|
|
|
|
VkDescriptorSet allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) {
|
|
VkDescriptorSet set = VK_NULL_HANDLE;
|
|
VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO};
|
|
desc_alloc.descriptorPool = descriptor_pool_;
|
|
desc_alloc.descriptorSetCount = 1;
|
|
desc_alloc.pSetLayouts = &descriptor_layout_;
|
|
check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui");
|
|
|
|
VkDescriptorImageInfo image_infos[2]{};
|
|
image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_;
|
|
image_infos[0].imageView = view0;
|
|
image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_);
|
|
image_infos[1].imageView = view1;
|
|
image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
|
|
VkWriteDescriptorSet writes[2]{};
|
|
for (int b = 0; b < 2; ++b) {
|
|
writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
|
|
writes[b].dstSet = set;
|
|
writes[b].dstBinding = static_cast<std::uint32_t>(b);
|
|
writes[b].descriptorCount = 1;
|
|
writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER;
|
|
writes[b].pImageInfo = &image_infos[b];
|
|
}
|
|
vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr);
|
|
return set;
|
|
}
|
|
|
|
GpuTexture& get_gpu_texture(
|
|
const std::string& texture_name,
|
|
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures) {
|
|
if (texture_name.empty()) return fallback_texture_;
|
|
if (texture_name.rfind("rt:", 0) == 0) {
|
|
RenderTarget* target = get_render_target(texture_name);
|
|
return target ? target->texture : fallback_texture_;
|
|
}
|
|
auto [it, inserted] = textures_.try_emplace(texture_name);
|
|
it->second.last_used_frame = frame_number_;
|
|
if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60))
|
|
return it->second;
|
|
if (inserted) {
|
|
it->second.descriptor = fallback_texture_.descriptor;
|
|
it->second.ui_descriptor = fallback_texture_.ui_descriptor;
|
|
}
|
|
it->second.last_decode_attempt_frame = frame_number_;
|
|
|
|
mtimage::Image decoded;
|
|
if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) {
|
|
decoded = decode_texture_bytes(raw->second.data(), raw->second.size());
|
|
}
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
if (!decoded.ok()) {
|
|
if (texture_name.rfind("mem:", 0) == 0) {
|
|
UIMemoryTexture mem_tex;
|
|
if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) {
|
|
const auto mtra = native_draw_capture::encode_raw_argb_as_mtra(
|
|
static_cast<std::uint32_t>(mem_tex.width),
|
|
static_cast<std::uint32_t>(mem_tex.height),
|
|
mem_tex.argb.data());
|
|
decoded = decode_texture_bytes(mtra.data(), mtra.size());
|
|
}
|
|
} else {
|
|
std::vector<std::uint8_t> pack_bytes;
|
|
if (read_live_pack_texture(texture_name, pack_bytes))
|
|
decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size());
|
|
}
|
|
}
|
|
#endif
|
|
if (!decoded.ok()) return it->second;
|
|
// 40250 CGraphicImageTexture: DXT DDS -> CreateDDSTexture with the file's levels;
|
|
// other files -> D3DXCreateTextureFromFileInMemoryEx(D3DX_DEFAULT) = full chain;
|
|
// memory images (mem:, CreateTexture(w, h, 1, ...)) -> one level.
|
|
const bool is_memory_tex = texture_name.rfind("mem:", 0) == 0;
|
|
last_texture_upload_name_ = texture_name;
|
|
upload_texture(
|
|
decoded.w, decoded.h, decoded.rgba.data(), it->second, true,
|
|
!is_memory_tex && decoded.file_mip_levels == 0, &decoded.mips);
|
|
it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view);
|
|
it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view);
|
|
return it->second;
|
|
}
|
|
|
|
std::uint64_t sampler_state_key(const Render3DDraw& draw, int stage) const {
|
|
return (std::uint64_t(draw.address_u[stage] & 15u)) |
|
|
(std::uint64_t(draw.address_v[stage] & 15u) << 4) |
|
|
(std::uint64_t(draw.min_filter[stage] & 15u) << 8) |
|
|
(std::uint64_t(draw.mag_filter[stage] & 15u) << 12) |
|
|
(std::uint64_t(draw.mip_filter[stage] & 15u) << 16);
|
|
}
|
|
|
|
VkSampler sampler_for_draw(const Render3DDraw& draw, int stage) {
|
|
const auto key = sampler_state_key(draw, stage);
|
|
if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_;
|
|
if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second;
|
|
auto address_mode = [](std::uint32_t mode) {
|
|
switch (mode) {
|
|
case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT;
|
|
case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE;
|
|
case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER;
|
|
default: return VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
|
}
|
|
};
|
|
VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO};
|
|
info.addressModeU = address_mode(draw.address_u[stage]);
|
|
info.addressModeV = address_mode(draw.address_v[stage]);
|
|
info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT;
|
|
info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE;
|
|
info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR;
|
|
info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR;
|
|
info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR;
|
|
info.minLod = 0.0f;
|
|
info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE;
|
|
if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) {
|
|
info.anisotropyEnable = VK_TRUE;
|
|
info.maxAnisotropy = max_anisotropy_;
|
|
}
|
|
VkSampler sampler = VK_NULL_HANDLE;
|
|
check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state");
|
|
state_samplers_.emplace(key, sampler);
|
|
return sampler;
|
|
}
|
|
|
|
VkDescriptorSet get_texture_descriptor(
|
|
const std::string& texture0_name,
|
|
const std::string& mask_name,
|
|
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures,
|
|
const Render3DDraw& draw) {
|
|
const auto& tex0 = get_gpu_texture(texture0_name, capture_textures);
|
|
const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures);
|
|
const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" +
|
|
std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1));
|
|
if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) {
|
|
it->second.last_used_frame = frame_number_;
|
|
return it->second.set;
|
|
}
|
|
const VkDescriptorSet set = allocate_texture_descriptor_set(
|
|
tex0.view ? tex0.view : fallback_texture_.view,
|
|
tex1.view ? tex1.view : fallback_texture_.view,
|
|
sampler_for_draw(draw, 0), sampler_for_draw(draw, 1));
|
|
paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_});
|
|
return set;
|
|
}
|
|
|
|
VkDescriptorSet get_ui_texture_descriptor(
|
|
const std::string& texture0_name,
|
|
const std::string& mask_name,
|
|
const std::unordered_map<std::string, std::vector<std::uint8_t>>& capture_textures) {
|
|
const auto& tex0 = get_gpu_texture(texture0_name, capture_textures);
|
|
if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor;
|
|
const auto& tex1 = get_gpu_texture(mask_name, capture_textures);
|
|
if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor;
|
|
const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name;
|
|
if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) {
|
|
it->second.last_used_frame = frame_number_;
|
|
return it->second.set;
|
|
}
|
|
const VkDescriptorSet set = allocate_ui_texture_descriptor_set(
|
|
tex0.view ? tex0.view : fallback_texture_.view,
|
|
tex1.view ? tex1.view : fallback_texture_.view);
|
|
paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_});
|
|
return set;
|
|
}
|
|
|
|
void upload_texture(
|
|
std::uint32_t width,
|
|
std::uint32_t height,
|
|
const std::uint8_t* rgba,
|
|
GpuTexture& texture,
|
|
bool count_upload,
|
|
bool generate_mips = true,
|
|
const std::vector<std::vector<std::uint8_t>>* file_mips = nullptr) {
|
|
const ScopedMs timer(timings_.texture_upload_ms);
|
|
const bool has_file_mips = !generate_mips && file_mips && !file_mips->empty();
|
|
VkDeviceSize byte_size = VkDeviceSize(width) * height * 4;
|
|
if (has_file_mips)
|
|
for (const auto& level : *file_mips) byte_size += level.size();
|
|
const std::uint32_t mip_levels = generate_mips
|
|
? (static_cast<std::uint32_t>(std::floor(std::log2(std::max(width, height)))) + 1u)
|
|
: has_file_mips ? 1u + static_cast<std::uint32_t>(file_mips->size()) : 1u;
|
|
// Inside a frame the copy is recorded ahead of that frame's render pass and reads the slot's
|
|
// staging ring; the GPU runs it with the frame, no queue wait. Outside a frame (the fallback
|
|
// texture at startup) it is a one-off submit that waits.
|
|
const bool in_frame = f_->recording;
|
|
VkBuffer staging_buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory staging_memory = VK_NULL_HANDLE;
|
|
VkDeviceSize staging_offset = 0;
|
|
std::uint8_t* mapped = nullptr;
|
|
const VkDeviceSize aligned_offset =
|
|
(f_->staging_used + staging_alignment_ - 1) / staging_alignment_ * staging_alignment_;
|
|
if (in_frame && f_->staging_mapped && aligned_offset + byte_size <= f_->staging_capacity) {
|
|
staging_buffer = f_->staging_buffer;
|
|
staging_offset = aligned_offset;
|
|
mapped = f_->staging_mapped + staging_offset;
|
|
f_->staging_used = aligned_offset + byte_size;
|
|
} else {
|
|
if (in_frame) f_->staging_wanted = std::max(f_->staging_wanted, aligned_offset + byte_size);
|
|
void* raw = nullptr;
|
|
create_host_buffer(byte_size, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, staging_buffer, staging_memory, &raw);
|
|
mapped = static_cast<std::uint8_t*>(raw);
|
|
}
|
|
std::memcpy(mapped, rgba, VkDeviceSize(width) * height * 4);
|
|
if (has_file_mips) {
|
|
auto* dst = mapped + VkDeviceSize(width) * height * 4;
|
|
for (const auto& level : *file_mips) { std::memcpy(dst, level.data(), level.size()); dst += level.size(); }
|
|
}
|
|
if (staging_memory) vkUnmapMemory(device_, staging_memory);
|
|
|
|
VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO};
|
|
image_info.imageType = VK_IMAGE_TYPE_2D;
|
|
image_info.format = VK_FORMAT_R8G8B8A8_UNORM;
|
|
image_info.extent = {width, height, 1};
|
|
image_info.mipLevels = mip_levels;
|
|
image_info.arrayLayers = 1;
|
|
image_info.samples = VK_SAMPLE_COUNT_1_BIT;
|
|
image_info.tiling = VK_IMAGE_TILING_OPTIMAL;
|
|
image_info.usage =
|
|
VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT;
|
|
image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture");
|
|
VkMemoryRequirements image_reqs{};
|
|
vkGetImageMemoryRequirements(device_, texture.image, &image_reqs);
|
|
VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
image_alloc.allocationSize = image_reqs.size;
|
|
image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
|
check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture");
|
|
check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture");
|
|
|
|
VkCommandBuffer upload_cmd = f_->command;
|
|
if (!in_frame) {
|
|
VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO};
|
|
cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1;
|
|
check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture");
|
|
VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO};
|
|
begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT;
|
|
check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture");
|
|
}
|
|
|
|
VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
|
|
to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED;
|
|
to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
|
to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
to_transfer.image = texture.image;
|
|
to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1};
|
|
vkCmdPipelineBarrier(
|
|
upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT,
|
|
0, 0, nullptr, 0, nullptr, 1, &to_transfer);
|
|
|
|
std::vector<VkBufferImageCopy> regions(has_file_mips ? mip_levels : 1u);
|
|
VkDeviceSize region_offset = 0;
|
|
for (std::uint32_t i = 0; i < regions.size(); ++i) {
|
|
const std::uint32_t lw = std::max(1u, width >> i), lh = std::max(1u, height >> i);
|
|
regions[i].bufferOffset = staging_offset + region_offset;
|
|
regions[i].imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1};
|
|
regions[i].imageExtent = {lw, lh, 1};
|
|
region_offset += VkDeviceSize(lw) * lh * 4;
|
|
}
|
|
vkCmdCopyBufferToImage(
|
|
upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
|
|
static_cast<std::uint32_t>(regions.size()), regions.data());
|
|
|
|
std::int32_t mip_w = static_cast<std::int32_t>(width);
|
|
std::int32_t mip_h = static_cast<std::int32_t>(height);
|
|
for (std::uint32_t i = 1; generate_mips && i < mip_levels; ++i) {
|
|
VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
|
|
barrier.image = texture.image;
|
|
barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1};
|
|
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
|
barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
|
|
barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
|
|
vkCmdPipelineBarrier(
|
|
upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT,
|
|
0, 0, nullptr, 0, nullptr, 1, &barrier);
|
|
|
|
const std::int32_t next_w = std::max(1, mip_w / 2);
|
|
const std::int32_t next_h = std::max(1, mip_h / 2);
|
|
VkImageBlit blit{};
|
|
blit.srcOffsets[0] = {0, 0, 0};
|
|
blit.srcOffsets[1] = {mip_w, mip_h, 1};
|
|
blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1};
|
|
blit.dstOffsets[0] = {0, 0, 0};
|
|
blit.dstOffsets[1] = {next_w, next_h, 1};
|
|
blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1};
|
|
vkCmdBlitImage(
|
|
upload_cmd,
|
|
texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
|
|
texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
|
|
1, &blit, VK_FILTER_LINEAR);
|
|
|
|
barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
|
|
barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT;
|
|
barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
|
|
vkCmdPipelineBarrier(
|
|
upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT,
|
|
0, 0, nullptr, 0, nullptr, 1, &barrier);
|
|
|
|
mip_w = next_w;
|
|
mip_h = next_h;
|
|
}
|
|
|
|
VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER};
|
|
to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
|
|
to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
|
|
to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
|
|
to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
|
|
to_shader.image = texture.image;
|
|
to_shader.subresourceRange = generate_mips
|
|
? VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1}
|
|
: VkImageSubresourceRange{VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1};
|
|
vkCmdPipelineBarrier(
|
|
upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT,
|
|
0, 0, nullptr, 0, nullptr, 1, &to_shader);
|
|
|
|
if (in_frame) {
|
|
if (staging_memory) f_->retired_buffers.emplace_back(staging_buffer, staging_memory);
|
|
} else {
|
|
check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture");
|
|
VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO};
|
|
submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd;
|
|
check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture");
|
|
check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture");
|
|
vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd);
|
|
vkDestroyBuffer(device_, staging_buffer, nullptr);
|
|
vkFreeMemory(device_, staging_memory, nullptr);
|
|
}
|
|
|
|
VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO};
|
|
view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D;
|
|
view_info.format = VK_FORMAT_R8G8B8A8_UNORM;
|
|
view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1};
|
|
check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture");
|
|
|
|
if (count_upload) {
|
|
++texture_upload_count_;
|
|
texture_uploaded_bytes_ += byte_size;
|
|
}
|
|
}
|
|
|
|
void release_texture(GpuTexture& texture) {
|
|
for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor})
|
|
if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor)
|
|
vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set);
|
|
if (texture.view) vkDestroyImageView(device_, texture.view, nullptr);
|
|
if (texture.image) vkDestroyImage(device_, texture.image, nullptr);
|
|
if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr);
|
|
texture = {};
|
|
}
|
|
|
|
void prune_textures() {
|
|
constexpr std::uint64_t static_retention_frames = 120;
|
|
constexpr std::uint64_t memory_retention_frames = 2;
|
|
for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) {
|
|
const auto retention = it->first.find("mem:") != std::string::npos
|
|
? memory_retention_frames : static_retention_frames;
|
|
if (frame_number_ - it->second.last_used_frame > retention) {
|
|
vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set);
|
|
it = paired_descriptors_.erase(it);
|
|
} else ++it;
|
|
}
|
|
for (auto it = textures_.begin(); it != textures_.end();) {
|
|
const auto retention = it->first.rfind("mem:", 0) == 0
|
|
? memory_retention_frames : static_retention_frames;
|
|
if (frame_number_ - it->second.last_used_frame > retention) {
|
|
release_texture(it->second);
|
|
it = textures_.erase(it);
|
|
} else ++it;
|
|
}
|
|
}
|
|
|
|
void release_geometry(Geometry& geometry) {
|
|
if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr);
|
|
if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr);
|
|
geometry = {};
|
|
}
|
|
|
|
void upload_geometry(const Render3DDraw& draw, Geometry& geometry) {
|
|
const ScopedMs timer(timings_.geometry_upload_ms);
|
|
if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3))
|
|
throw std::runtime_error("invalid native geometry array size");
|
|
const auto count = draw.positions.size() / 3;
|
|
if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX)
|
|
throw std::runtime_error("geometry exceeds Vulkan index range");
|
|
// D3D8 feeds opaque white for a vertex format without D3DFVF_DIFFUSE.
|
|
const std::uint32_t default_argb = 0xffffffffu;
|
|
const bool has_skin =
|
|
draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4;
|
|
std::vector<Vertex> vertices(count);
|
|
for (std::size_t i = 0; i < count; ++i) {
|
|
vertices[i].position[0] = draw.positions[3*i];
|
|
vertices[i].position[1] = draw.positions[3*i+1];
|
|
vertices[i].position[2] = draw.positions[3*i+2];
|
|
vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f;
|
|
vertices[i].vertex_fog = i < draw.vertex_fog.size() ? draw.vertex_fog[i] : 1.0f;
|
|
if (3*i + 2 < draw.normals.size()) {
|
|
vertices[i].normal[0] = draw.normals[3*i];
|
|
vertices[i].normal[1] = draw.normals[3*i+1];
|
|
vertices[i].normal[2] = draw.normals[3*i+2];
|
|
} else {
|
|
vertices[i].normal[0] = 0.0f;
|
|
vertices[i].normal[1] = 0.0f;
|
|
vertices[i].normal[2] = 1.0f;
|
|
}
|
|
if (2*i + 1 < draw.uv0.size()) {
|
|
vertices[i].uv[0] = draw.uv0[2*i];
|
|
vertices[i].uv[1] = draw.uv0[2*i+1];
|
|
} else {
|
|
vertices[i].uv[0] = 0.0f;
|
|
vertices[i].uv[1] = 0.0f;
|
|
}
|
|
const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb;
|
|
vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f;
|
|
vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f;
|
|
vertices[i].color[2] = float(argb & 255) / 255.0f;
|
|
vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f;
|
|
if (has_skin) {
|
|
for (int k = 0; k < 4; ++k) {
|
|
vertices[i].joints[k] = draw.bone_indices[4*i + k];
|
|
vertices[i].weights[k] = draw.bone_weights[4*i + k];
|
|
}
|
|
} else {
|
|
vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0;
|
|
vertices[i].weights[0] = 1.0f;
|
|
vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f;
|
|
}
|
|
if (2*i + 1 < draw.uv1.size()) {
|
|
vertices[i].mask_uv[0] = draw.uv1[2*i];
|
|
vertices[i].mask_uv[1] = draw.uv1[2*i+1];
|
|
} else {
|
|
vertices[i].mask_uv[0] = 0.0f;
|
|
vertices[i].mask_uv[1] = 0.0f;
|
|
}
|
|
}
|
|
for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count");
|
|
geometry.index_offset = vertices.size() * sizeof(Vertex);
|
|
const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t);
|
|
if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB");
|
|
VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO};
|
|
info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT;
|
|
info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
|
|
check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer");
|
|
VkMemoryRequirements requirements{};
|
|
vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements);
|
|
VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO};
|
|
allocation.allocationSize = requirements.size;
|
|
allocation.memoryTypeIndex = find_memory_type(
|
|
requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT);
|
|
check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory");
|
|
check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory");
|
|
void* mapped = nullptr;
|
|
check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory");
|
|
std::memcpy(mapped, vertices.data(), geometry.index_offset);
|
|
std::memcpy(static_cast<std::uint8_t*>(mapped) + geometry.index_offset,
|
|
draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t));
|
|
vkUnmapMemory(device_, geometry.memory);
|
|
geometry.vertex_count = static_cast<std::uint32_t>(count);
|
|
geometry.index_count = static_cast<std::uint32_t>(draw.indices.size());
|
|
++upload_count_;
|
|
uploaded_bytes_ += bytes;
|
|
}
|
|
|
|
bool vsync_ = true;
|
|
VkPresentModeKHR present_mode_ = VK_PRESENT_MODE_FIFO_KHR;
|
|
float timestamp_period_ns_ = 1.0f;
|
|
float max_anisotropy_ = 1.0f;
|
|
SDL_Window* window_ = nullptr;
|
|
unsigned screen_keyboard_serial_ = 0;
|
|
int safe_inset_px_ = 0;
|
|
bool show_fps_ = false;
|
|
int fps_ = -1;
|
|
int fps_frames_ = 0;
|
|
std::chrono::steady_clock::time_point fps_window_start_{};
|
|
|
|
// Adds the scope's wall time to a Timings field.
|
|
struct ScopedMs {
|
|
explicit ScopedMs(double& into) : into_(into), start_(std::chrono::steady_clock::now()) {}
|
|
~ScopedMs() { into_ += std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now() - start_).count(); }
|
|
double& into_;
|
|
std::chrono::steady_clock::time_point start_;
|
|
};
|
|
|
|
// Presented frames over the last half second.
|
|
void update_fps_counter() {
|
|
const auto now = std::chrono::steady_clock::now();
|
|
if (fps_frames_++ == 0) {
|
|
fps_window_start_ = now;
|
|
return;
|
|
}
|
|
const double elapsed = std::chrono::duration<double>(now - fps_window_start_).count();
|
|
if (elapsed >= 0.5) {
|
|
fps_ = int(std::lround((fps_frames_ - 1) / elapsed));
|
|
fps_frames_ = 1;
|
|
fps_window_start_ = now;
|
|
}
|
|
}
|
|
VkInstance instance_ = VK_NULL_HANDLE;
|
|
VkSurfaceKHR surface_ = VK_NULL_HANDLE;
|
|
VkPhysicalDevice physical_ = VK_NULL_HANDLE;
|
|
VkDevice device_ = VK_NULL_HANDLE;
|
|
VkQueue queue_ = VK_NULL_HANDLE;
|
|
std::uint32_t queue_family_ = 0;
|
|
std::string device_name_;
|
|
VkSwapchainKHR swapchain_ = VK_NULL_HANDLE;
|
|
VkFormat swapchain_format_ = VK_FORMAT_UNDEFINED;
|
|
VkExtent2D extent_{};
|
|
std::vector<VkImageView> image_views_;
|
|
std::vector<VkImage> swapchain_images_;
|
|
// --screenshot-out: read back the swapchain image of one frame into a BMP file.
|
|
std::string screenshot_path_;
|
|
std::uint64_t screenshot_frame_ = 0;
|
|
VkImage depth_image_ = VK_NULL_HANDLE;
|
|
VkDeviceMemory depth_memory_ = VK_NULL_HANDLE;
|
|
VkImageView depth_view_ = VK_NULL_HANDLE;
|
|
VkSampleCountFlagBits samples_ = VK_SAMPLE_COUNT_1_BIT;
|
|
VkImage msaa_image_ = VK_NULL_HANDLE;
|
|
VkDeviceMemory msaa_memory_ = VK_NULL_HANDLE;
|
|
VkImageView msaa_view_ = VK_NULL_HANDLE;
|
|
VkRenderPass pass_ = VK_NULL_HANDLE;
|
|
std::vector<VkFramebuffer> framebuffers_;
|
|
VkSampler sampler_ = VK_NULL_HANDLE;
|
|
VkSampler clamp_sampler_ = VK_NULL_HANDLE;
|
|
VkSampler ui_sampler_ = VK_NULL_HANDLE;
|
|
std::unordered_map<std::uint64_t, VkSampler> state_samplers_;
|
|
VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE;
|
|
VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE;
|
|
VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE;
|
|
// Everything one frame writes while the GPU may still read the previous frame's copy: with
|
|
// kFramesInFlight slots the CPU prepares frame N while the GPU draws frame N-1. A slot is reused
|
|
// only after its fence (the submit of frame N - frames_in_flight_) has signalled.
|
|
struct FrameSlot {
|
|
VkCommandBuffer command = VK_NULL_HANDLE;
|
|
VkFence fence = VK_NULL_HANDLE;
|
|
VkSemaphore acquire = VK_NULL_HANDLE;
|
|
std::uint32_t query_base = 0;
|
|
bool has_pending_query = false;
|
|
bool recording = false;
|
|
VkDescriptorSet bone_descriptor_set = VK_NULL_HANDLE;
|
|
VkBuffer bone_buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory bone_memory = VK_NULL_HANDLE;
|
|
float* bone_mapped = nullptr;
|
|
std::size_t bone_capacity = 0;
|
|
VkBuffer state_buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory state_memory = VK_NULL_HANDLE;
|
|
std::uint8_t* state_mapped = nullptr;
|
|
std::size_t state_capacity = 0;
|
|
VkBuffer ui_buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory ui_memory = VK_NULL_HANDLE;
|
|
std::uint8_t* ui_mapped = nullptr;
|
|
std::size_t ui_vertex_capacity = 0, ui_index_capacity = 0;
|
|
VkDeviceSize ui_vertex_bytes = 0;
|
|
// Texture uploads recorded into this frame's command buffer copy from here.
|
|
VkBuffer staging_buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory staging_memory = VK_NULL_HANDLE;
|
|
std::uint8_t* staging_mapped = nullptr;
|
|
VkDeviceSize staging_capacity = 0, staging_used = 0, staging_wanted = 0;
|
|
// Uploads larger than the ring get their own staging buffer, freed when the slot comes back.
|
|
std::vector<std::pair<VkBuffer, VkDeviceMemory>> retired_buffers;
|
|
};
|
|
static constexpr std::uint32_t kMaxFramesInFlight = 2;
|
|
std::array<FrameSlot, kMaxFramesInFlight> frames_{};
|
|
FrameSlot* f_ = &frames_[0];
|
|
std::uint32_t frames_in_flight_ = kMaxFramesInFlight;
|
|
std::uint32_t frame_slot_ = 0;
|
|
// Signalled by a frame's submit, waited by the present of that swapchain image.
|
|
std::vector<VkSemaphore> rendered_;
|
|
VkDeviceSize state_stride_ = 0;
|
|
VkDeviceSize staging_alignment_ = 16;
|
|
GpuTexture fallback_texture_{};
|
|
std::unordered_map<std::string, GpuTexture> textures_;
|
|
std::unordered_map<std::string, PairedDescriptor> paired_descriptors_;
|
|
std::unordered_map<std::string, RenderTarget> render_targets_;
|
|
VkRenderPass offscreen_pass_ = VK_NULL_HANDLE;
|
|
VkFormat offscreen_format_ = VK_FORMAT_R5G6B5_UNORM_PACK16;
|
|
std::size_t last_offscreen_draw_count_ = 0;
|
|
VkShaderModule vertex_module_ = VK_NULL_HANDLE;
|
|
VkShaderModule fragment_module_ = VK_NULL_HANDLE;
|
|
VkPipelineLayout layout_ = VK_NULL_HANDLE;
|
|
std::unordered_map<std::uint32_t, VkPipeline> pipelines_;
|
|
std::unordered_map<GeometryId, Geometry, GeometryIdHash> geometries_;
|
|
std::uint64_t frame_number_ = 0;
|
|
std::uint64_t timed_frames_ = 0;
|
|
std::size_t upload_count_ = 0, uploaded_bytes_ = 0;
|
|
std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0;
|
|
std::string last_texture_upload_name_;
|
|
VkCommandPool pool_ = VK_NULL_HANDLE;
|
|
VkQueryPool query_pool_ = VK_NULL_HANDLE;
|
|
bool os_cursor_hidden_ = false;
|
|
bool hardware_cursor_enabled_ = false;
|
|
std::array<SDL_Cursor*, 15> game_cursors_{};
|
|
SDL_Cursor* fallback_cursor_ = nullptr;
|
|
int current_cursor_shape_ = -1;
|
|
int last_mx_ = 0, last_my_ = 0;
|
|
std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0;
|
|
std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0;
|
|
std::size_t last_vertex_count_ = 0, last_index_count_ = 0;
|
|
Timings timings_{};
|
|
double last_sync_ms_ = 0;
|
|
};
|
|
|
|
std::vector<Render3DDraw> make_test_draws(std::size_t count, std::size_t triangles_per_draw) {
|
|
std::vector<Render3DDraw> draws;
|
|
draws.reserve(count);
|
|
for (std::size_t i = 0; i < count; ++i) {
|
|
Render3DDraw draw{};
|
|
draw.geometry_key = i + 1;
|
|
draw.geometry_revision = 1;
|
|
draw.z_enable = 1;
|
|
draw.z_write = 1;
|
|
for (auto* matrix : {draw.world, draw.view, draw.proj})
|
|
for (int diagonal = 0; diagonal < 4; ++diagonal) matrix[diagonal * 5] = 1.0f;
|
|
const float x = (float(i % 8) - 3.5f) * 0.24f;
|
|
const float y = (float(i / 8) - 3.5f) * 0.22f;
|
|
draw.positions.reserve(triangles_per_draw * 9);
|
|
draw.indices.reserve(triangles_per_draw * 3);
|
|
draw.diffuse.reserve(triangles_per_draw * 3);
|
|
for (std::size_t t = 0; t < triangles_per_draw; ++t) {
|
|
const float tx = x + (float(t % 16) - 7.5f) * 0.009f;
|
|
const float ty = y + (float(t / 16) - float(triangles_per_draw / 32)) * 0.006f;
|
|
const auto base = static_cast<std::uint32_t>(draw.positions.size() / 3);
|
|
draw.positions.insert(draw.positions.end(), {tx-0.004f, ty-0.004f, 0.5f,
|
|
tx+0.004f, ty-0.004f, 0.5f,
|
|
tx, ty+0.004f, 0.5f});
|
|
draw.indices.insert(draw.indices.end(), {base, base+1, base+2});
|
|
draw.diffuse.insert(draw.diffuse.end(), {0xff4fbfff, 0xff4fbfff, 0xff4fbfff});
|
|
}
|
|
draws.push_back(std::move(draw));
|
|
}
|
|
return draws;
|
|
}
|
|
|
|
void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms = 0.0,
|
|
std::vector<double> frame_ms = {}) {
|
|
const auto timing = renderer.timings();
|
|
std::sort(frame_ms.begin(), frame_ms.end());
|
|
const auto percentile = [&](double fraction) {
|
|
if (frame_ms.empty()) return 0.0;
|
|
const auto index = static_cast<std::size_t>(std::ceil(fraction * frame_ms.size())) - 1;
|
|
return frame_ms[std::min(index, frame_ms.size() - 1)];
|
|
};
|
|
std::cout << "device=" << renderer.device_name()
|
|
<< " present_mode=" << renderer.present_mode_name() << " msaa=" << renderer.msaa_samples()
|
|
<< " frames=" << completed
|
|
<< " draws=" << renderer.draw_count()
|
|
<< " skinned_draws=" << renderer.skinned_draw_count()
|
|
<< " ui_batches=" << renderer.ui_batch_count()
|
|
<< " ui_quads=" << renderer.ui_quad_count()
|
|
<< " vertices=" << renderer.vertex_count()
|
|
<< " indices=" << renderer.index_count()
|
|
<< " uploads=" << renderer.upload_count()
|
|
<< " uploaded_bytes=" << renderer.uploaded_bytes()
|
|
<< " texture_uploads=" << renderer.texture_upload_count()
|
|
<< " texture_bytes=" << renderer.texture_uploaded_bytes()
|
|
<< " mean_frame_ms=" << (completed ? wall_ms / completed : 0.0)
|
|
<< " p95_frame_ms=" << percentile(0.95)
|
|
<< " p99_frame_ms=" << percentile(0.99)
|
|
<< " max_frame_ms=" << percentile(1.0)
|
|
<< " game_update_ms=" << (completed ? update_ms / completed : 0.0)
|
|
<< " sync_ms=" << (completed ? timing.sync_ms / completed : 0.0)
|
|
<< " prepare_ms=" << (completed ? timing.prepare_ms / completed : 0.0)
|
|
<< " steady_prepare_ms=" << (completed > 1 ? timing.steady_prepare_ms / (completed - 1) : 0.0)
|
|
<< " submit_ms=" << (completed ? timing.submit_ms / completed : 0.0)
|
|
<< " present_ms=" << (completed ? timing.present_ms / completed : 0.0)
|
|
<< " gpu_ms=" << (timing.gpu_samples ? timing.gpu_ms / double(timing.gpu_samples) : 0.0) << '\n';
|
|
}
|
|
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
// --py-exec: a test hook run once the auto-login run reaches the GameWindow (or, with --login-screen,
|
|
// once the LoginWindow is up).
|
|
std::string g_py_exec_after_login;
|
|
// --perf-log / MT_PERF_LOG=1: native_render/perf_log.h.
|
|
bool g_perf_log = false;
|
|
// --fps-cap N: the loop starts a frame at most every 1/N s (0 = as fast as vsync allows).
|
|
int g_fps_cap = 0;
|
|
// --refresh-rate N: the surface's preferred frame rate (Android ANativeWindow_setFrameRate; MainActivity
|
|
// also picks the matching display mode). 0 leaves the system default.
|
|
int g_refresh_rate = 0;
|
|
|
|
bool py_exec(const std::string& code) {
|
|
std::string err;
|
|
if (!PythonBoot::RunLine(code.c_str(), &err)) {
|
|
std::cerr << "PythonBoot::RunLine failed: " << err << '\n';
|
|
return false;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
bool py_eval_true(const char* expr) {
|
|
std::string res, err;
|
|
return PythonBoot::Evaluate(expr, &res, &err) && (res == "True" || res == "1");
|
|
}
|
|
|
|
#ifdef __ANDROID__
|
|
bool hide_mobile_taskbar_quickslots() {
|
|
return py_exec(
|
|
"_mobile_taskbar = _stream.curPhaseWindow.interface.wndTaskBar\n"
|
|
"for _mobile_quickslot in _mobile_taskbar.quickslot:\n"
|
|
" _mobile_quickslot.Hide()\n"
|
|
"_mobile_taskbar.GetChild('QuickSlotBoard').Hide()\n");
|
|
}
|
|
#endif
|
|
|
|
#if defined(__APPLE__) && TARGET_OS_OSX
|
|
bool configure_macos_numeric_quickslots() {
|
|
return py_exec(
|
|
"_mac_game = _stream.curPhaseWindow\n"
|
|
"_mac_game.onPressKeyDict[app.DIK_5] = lambda: _mac_game._GameWindow__PressQuickSlot(4)\n"
|
|
"_mac_game.onPressKeyDict[app.DIK_6] = lambda: _mac_game._GameWindow__PressQuickSlot(5)\n"
|
|
"_mac_game.onPressKeyDict[app.DIK_7] = lambda: _mac_game._GameWindow__PressQuickSlot(6)\n"
|
|
"_mac_game.onPressKeyDict[app.DIK_8] = lambda: _mac_game._GameWindow__PressQuickSlot(7)\n"
|
|
"for _mac_key in (app.DIK_F1, app.DIK_F2, app.DIK_F3, app.DIK_F4):\n"
|
|
" del _mac_game.onPressKeyDict[_mac_key]\n"
|
|
"_mac_taskbar = _mac_game.interface.wndTaskBar\n"
|
|
"for _mac_slot_number in range(5, 9):\n"
|
|
" _mac_taskbar.GetChild('slot_%d' % _mac_slot_number).LoadImage('d:/ymir work/ui/game/taskbar/%d.sub' % _mac_slot_number)\n");
|
|
}
|
|
#endif
|
|
|
|
bool parse_live_server_spec(const std::string& spec, std::string& host, int& auth_port, int& game_port) {
|
|
const auto p1 = spec.find(':');
|
|
if (p1 == std::string::npos) return false;
|
|
const auto p2 = spec.find(':', p1 + 1);
|
|
if (p2 == std::string::npos) return false;
|
|
host = spec.substr(0, p1);
|
|
auth_port = std::stoi(spec.substr(p1 + 1, p2 - p1 - 1));
|
|
game_port = std::stoi(spec.substr(p2 + 1));
|
|
return !host.empty() && auth_port > 0 && game_port > 0;
|
|
}
|
|
|
|
int run_live_client(
|
|
VulkanWindow& renderer,
|
|
const std::string& client_dir,
|
|
int frames,
|
|
int fake_mobs,
|
|
bool gpu_skinning,
|
|
bool native_terrain,
|
|
bool login_screen,
|
|
bool selection_screen,
|
|
const std::string& live_server_spec,
|
|
const std::string& capture_out,
|
|
const std::string& screenshot_out) {
|
|
if (fake_mobs > 0) {
|
|
const std::string mobs_str = std::to_string(fake_mobs);
|
|
setenv("MT_FAKE_MOB_COUNT", mobs_str.c_str(), 1);
|
|
}
|
|
#ifdef __ANDROID__
|
|
SetTraceErrorObserver([](const char* line) {
|
|
SDL_LogError(SDL_LOG_CATEGORY_APPLICATION, "40250 SYSERR: %s", line);
|
|
});
|
|
#endif
|
|
char resolved_client_dir[4096] = {};
|
|
const std::string abs_client_dir = realpath(client_dir.c_str(), resolved_client_dir) ? std::string(resolved_client_dir) : client_dir;
|
|
setenv("MT_40250_CLIENT", abs_client_dir.c_str(), 1);
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: pack init %s", abs_client_dir.c_str());
|
|
#endif
|
|
if (chdir(abs_client_dir.c_str()) != 0 || !mtpack40250::initialize(".")) {
|
|
throw std::runtime_error("cannot initialize 40250 pack at: " + abs_client_dir);
|
|
}
|
|
|
|
const bool host_hardware_cursor = !renderer.touch_controller.is_enabled();
|
|
PythonBoot::SetHostHardwareCursorEnabled(host_hardware_cursor);
|
|
if (host_hardware_cursor) renderer.enable_game_hardware_cursor();
|
|
|
|
SetNativeTerrainRenderEnabled(native_terrain);
|
|
SetGpuSkinningEnabled(gpu_skinning);
|
|
// The shadow map is drawn by the GPU (offscreen pass); MT_CPU_SHADOW=1 / --cpu-shadow keeps the
|
|
// CPU rasterizer (RecordingDevice::rasterize_shadow) for comparison.
|
|
const char* cpu_shadow = std::getenv("MT_CPU_SHADOW");
|
|
SetGpuRenderTargetsEnabled(!(cpu_shadow && *cpu_shadow == '1'));
|
|
NativeAudioEngine audio;
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: audio initialized");
|
|
#endif
|
|
|
|
std::string server_host = "127.0.0.1";
|
|
int auth_port = 0;
|
|
int game_port = 0;
|
|
const bool use_external_server = !live_server_spec.empty();
|
|
FakeLoginServer server;
|
|
if (use_external_server) {
|
|
if (!parse_live_server_spec(live_server_spec, server_host, auth_port, game_port))
|
|
throw std::runtime_error("invalid --live-server spec (expected HOST:AUTH_PORT:GAME_PORT): " + live_server_spec);
|
|
} else {
|
|
if (!server.Start()) throw std::runtime_error("FakeLoginServer failed to start on loopback");
|
|
auth_port = server.AuthPort();
|
|
game_port = server.GamePort();
|
|
}
|
|
|
|
{
|
|
// HiDPI text: rasterise glyph pages at the swapchain's pixels per UI pixel (before any font
|
|
// exists). Rounded up: a fractional density (Android 1760x800 UI on 2376x1080 = 1.35) then
|
|
// scales finer glyphs down instead of stretching 1x ones. MT_FONT_OVERSAMPLE=1 keeps the
|
|
// 40250 bilevel 1x glyphs.
|
|
const float density = float(renderer.width()) / float(std::max(1u, renderer.logical_width()));
|
|
int oversample = std::clamp(int(std::ceil(density - 0.05f)), 1, 4);
|
|
if (const char* env = std::getenv("MT_FONT_OVERSAMPLE")) oversample = std::max(1, std::atoi(env));
|
|
UISetFontOversample(oversample);
|
|
}
|
|
|
|
std::string error;
|
|
const char* env_stdlib = std::getenv("MT_PYTHON_STDLIB");
|
|
std::string stdlib = (env_stdlib && *env_stdlib) ? env_stdlib : bundled_file_path("python27.zip");
|
|
#ifdef __ANDROID__
|
|
if (!env_stdlib || !*env_stdlib) {
|
|
std::size_t zip_size = 0;
|
|
void* zip = SDL_LoadFile("assets://python27.zip", &zip_size);
|
|
if (!zip) zip = SDL_LoadFile("python27.zip", &zip_size);
|
|
if (!zip) throw std::runtime_error("cannot load bundled python27.zip");
|
|
char* pref = SDL_GetPrefPath("mtgodot", "native-render");
|
|
if (!pref) { SDL_free(zip); throw std::runtime_error("cannot locate app data directory"); }
|
|
stdlib = std::string(pref) + "python27.zip";
|
|
SDL_free(pref);
|
|
std::ofstream output(stdlib, std::ios::binary | std::ios::trunc);
|
|
output.write(static_cast<const char*>(zip), static_cast<std::streamsize>(zip_size));
|
|
SDL_free(zip);
|
|
if (!output) throw std::runtime_error("cannot extract bundled python27.zip");
|
|
}
|
|
#endif
|
|
if (!PythonBoot::Start(stdlib.c_str(), &error))
|
|
throw std::runtime_error("PythonBoot::Start: " + error);
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: PythonBoot started");
|
|
#endif
|
|
|
|
SetPlatformServerTime(123456789);
|
|
PythonBoot::SetUISafeInset(renderer.ui_safe_inset());
|
|
PythonBoot::SetUISize(static_cast<int>(renderer.logical_width()), static_cast<int>(renderer.logical_height()));
|
|
if (!PythonBoot::RunMainScript("", &error) || !PythonBoot::IsAppLooping())
|
|
throw std::runtime_error("PythonBoot::RunMainScript: " + error);
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: main script started");
|
|
#endif
|
|
|
|
const std::unordered_map<std::string, std::vector<std::uint8_t>> empty_textures;
|
|
auto pump_until = [&](double max_seconds, int sleep_ms, const auto& cond) -> bool {
|
|
const auto deadline = std::chrono::steady_clock::now() + std::chrono::duration<double>(max_seconds);
|
|
while (std::chrono::steady_clock::now() < deadline && renderer.poll(true) && PythonBoot::IsAppLooping()) {
|
|
PythonBoot::UIUpdate();
|
|
renderer.sync_game_cursor();
|
|
PythonBoot::UIRender();
|
|
audio.pump();
|
|
unsigned ui_w = renderer.width(), ui_h = renderer.height();
|
|
UIRenderGetSize(&ui_w, &ui_h);
|
|
renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands());
|
|
if (cond()) return true;
|
|
if (sleep_ms > 0)
|
|
std::this_thread::sleep_for(std::chrono::milliseconds(sleep_ms));
|
|
}
|
|
return cond();
|
|
};
|
|
|
|
if (!py_exec("import __main__, app, networkModule, introLogin, introSelect, introLoading, game\n"
|
|
"__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]"))
|
|
throw std::runtime_error("failed to locate networkModule.MainStream");
|
|
|
|
if (!pump_until(20.0, 5, [] { return py_eval_true("isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)"); }))
|
|
throw std::runtime_error("timed out waiting for LoginWindow");
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: LoginWindow open");
|
|
#endif
|
|
|
|
if (login_screen) {
|
|
char conn_cmd[512];
|
|
std::snprintf(
|
|
conn_cmd,
|
|
sizeof(conn_cmd),
|
|
"_stream.SetConnectInfo('%s', %d, '%s', %d)\n"
|
|
"_w = _stream.curPhaseWindow\n"
|
|
"_w._LoginWindow__OpenServerBoard()",
|
|
server_host.c_str(),
|
|
game_port,
|
|
server_host.c_str(),
|
|
auth_port);
|
|
if (!py_exec(conn_cmd))
|
|
throw std::runtime_error("failed to configure LoginWindow connection info");
|
|
for (int i = 0; i < 15; ++i) {
|
|
PythonBoot::UIUpdate();
|
|
renderer.sync_game_cursor();
|
|
PythonBoot::UIRender();
|
|
}
|
|
if (!g_py_exec_after_login.empty() && !py_exec(g_py_exec_after_login))
|
|
throw std::runtime_error("--py-exec failed");
|
|
} else {
|
|
char login_cmd[512];
|
|
std::snprintf(
|
|
login_cmd,
|
|
sizeof(login_cmd),
|
|
"_stream.SetConnectInfo('%s', %d, '%s', %d)\n"
|
|
"_w = _stream.curPhaseWindow\n"
|
|
"_w._LoginWindow__OpenLoginBoard()\n"
|
|
"_w.idEditLine.SetText('%s')\n"
|
|
"_w.pwdEditLine.SetText('%s')\n"
|
|
"_w._LoginWindow__OnClickLoginButton()",
|
|
server_host.c_str(),
|
|
game_port,
|
|
server_host.c_str(),
|
|
auth_port,
|
|
FakeLoginServer::kLogin,
|
|
FakeLoginServer::kPassword);
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: submitting fake login");
|
|
#endif
|
|
if (!py_exec(login_cmd))
|
|
throw std::runtime_error("failed to submit login credentials");
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: fake login submitted");
|
|
#endif
|
|
|
|
if (!pump_until(20.0, 5, [&] {
|
|
return (use_external_server || server.Has("game1:login2")) &&
|
|
py_eval_true("isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)");
|
|
}))
|
|
throw std::runtime_error("timed out waiting for SelectCharacterWindow");
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: SelectCharacterWindow open");
|
|
#endif
|
|
|
|
if (!selection_screen) {
|
|
if (!py_exec("_stream.curPhaseWindow.SelectSlot(0)\n"
|
|
"_stream.curPhaseWindow.StartGame()"))
|
|
throw std::runtime_error("failed to start game from SelectCharacterWindow");
|
|
|
|
if (!use_external_server) {
|
|
if (!pump_until(30.0, 5, [&] { return server.Has("game2:client_version"); }))
|
|
throw std::runtime_error("timed out waiting for LoadingWindow (client_version)");
|
|
}
|
|
|
|
if (!pump_until(60.0, 0, [&] {
|
|
return (use_external_server || server.Has("game2:burst_pong")) &&
|
|
py_eval_true("isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()");
|
|
}))
|
|
throw std::runtime_error("timed out waiting for GameWindow (server error: " + server.Error() + ")");
|
|
|
|
#ifdef __ANDROID__
|
|
if (!hide_mobile_taskbar_quickslots())
|
|
throw std::runtime_error("failed to hide the desktop quick-slot bar on Android");
|
|
#endif
|
|
#if defined(__APPLE__) && TARGET_OS_OSX
|
|
if (!configure_macos_numeric_quickslots())
|
|
throw std::runtime_error("failed to configure numeric quick slots on macOS");
|
|
#endif
|
|
if (!g_py_exec_after_login.empty() && !py_exec(g_py_exec_after_login))
|
|
throw std::runtime_error("--py-exec failed");
|
|
|
|
const int warmup_frames = std::max(10, fake_mobs / 2 + 10);
|
|
for (int i = 0; i < warmup_frames; ++i) {
|
|
if (!renderer.poll(true) || !PythonBoot::IsAppLooping()) break;
|
|
PythonBoot::UIUpdate();
|
|
renderer.sync_game_cursor();
|
|
audio.pump();
|
|
}
|
|
}
|
|
}
|
|
|
|
// Render one warmup frame to upload initial scene & UI textures/geometries before timed benchmark.
|
|
{
|
|
unsigned ui_w = renderer.width(), ui_h = renderer.height();
|
|
UIRenderGetSize(&ui_w, &ui_h);
|
|
renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands());
|
|
}
|
|
|
|
if (!capture_out.empty()) {
|
|
std::unordered_map<std::string, std::vector<std::uint8_t>> cap_textures;
|
|
auto add_tex = [&](const std::string& name) {
|
|
if (name.empty() || cap_textures.count(name)) return;
|
|
if (name.rfind("mem:", 0) == 0) {
|
|
UIMemoryTexture mem_tex;
|
|
if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) {
|
|
auto mtra = native_draw_capture::encode_raw_argb_as_mtra(
|
|
static_cast<std::uint32_t>(mem_tex.width),
|
|
static_cast<std::uint32_t>(mem_tex.height),
|
|
mem_tex.argb.data());
|
|
if (!mtra.empty()) cap_textures.emplace(name, std::move(mtra));
|
|
}
|
|
return;
|
|
}
|
|
std::vector<std::uint8_t> bytes;
|
|
if (read_live_pack_texture(name, bytes))
|
|
cap_textures.emplace(name, std::move(bytes));
|
|
};
|
|
for (const auto& d : Render3DDraws()) {
|
|
add_tex(d.texture0);
|
|
add_tex(d.texture1);
|
|
}
|
|
for (const auto& c : UIRenderCommands()) {
|
|
if (c.kind == UIRenderCommand::Image) {
|
|
add_tex(c.text);
|
|
add_tex(c.mask);
|
|
}
|
|
}
|
|
unsigned ui_w = renderer.width(), ui_h = renderer.height();
|
|
UIRenderGetSize(&ui_w, &ui_h);
|
|
std::vector<UIRenderCommand> cap_ui = UIRenderCommands();
|
|
if (renderer.touch_controller.is_enabled()) {
|
|
renderer.touch_controller.update_screen_size(int(ui_w), int(ui_h));
|
|
renderer.touch_controller.append_ui_commands(cap_ui);
|
|
}
|
|
native_draw_capture::write(capture_out, Render3DDraws(), cap_textures, ui_w, ui_h, cap_ui);
|
|
}
|
|
|
|
if (!screenshot_out.empty() && frames > 0) {
|
|
renderer.request_screenshot(screenshot_out, renderer.frame_number() + static_cast<std::uint64_t>(frames));
|
|
}
|
|
|
|
renderer.reset_timings();
|
|
native_perf::PerfLog perf_log;
|
|
if (g_perf_log) {
|
|
perf_log.open("device=" + renderer.device_name() + " present_mode=" + renderer.present_mode_name() +
|
|
" msaa=" + std::to_string(renderer.msaa_samples()) + " size=" + std::to_string(renderer.width()) +
|
|
"x" + std::to_string(renderer.height()) + " gpu_skinning=" + std::to_string(gpu_skinning) +
|
|
" terrain=" + std::to_string(native_terrain));
|
|
}
|
|
if (g_refresh_rate > 0) {
|
|
const bool ok = android_perf::request_frame_rate(renderer.sdl_window(), float(g_refresh_rate));
|
|
SDL_Log("frame rate request %d Hz: %s", g_refresh_rate, ok ? "accepted" : "unavailable");
|
|
}
|
|
perf_log.set_fps_cap(g_fps_cap);
|
|
perf_log.set_refresh_rate_source([] { return android_perf::display_refresh_rate(); });
|
|
perf_log.set_render_rate_source([] { return android_perf::render_rate(); });
|
|
perf_log.set_battery_source([] { return android_perf::battery_celsius(); });
|
|
// ADPF: each frame's CPU work (everything but the vsync/fence waits and the cap's sleep) against the
|
|
// frame budget. Opened once the script thread exists, i.e. here.
|
|
using steady = std::chrono::steady_clock;
|
|
// The target is 80% of the frame period: the governor settles the clocks so the reported work just
|
|
// meets the target, so a target equal to the period leaves every other frame late (cap 60 on the
|
|
// test phone: 54 fps against the period, 60 against 80% of it).
|
|
const auto frame_budget = [] {
|
|
const int fps = g_fps_cap > 0 ? g_fps_cap : (g_refresh_rate > 0 ? g_refresh_rate : 60);
|
|
return std::chrono::nanoseconds(800'000'000LL / fps);
|
|
}();
|
|
android_perf::PerformanceHint hint;
|
|
{
|
|
std::vector<int32_t> tids{android_perf::current_thread_id()};
|
|
if (const int script = PythonBoot::ScriptThreadId()) tids.push_back(script);
|
|
const bool ok = hint.open(tids, frame_budget.count());
|
|
SDL_Log("ADPF performance hint (%zu threads, %.2f ms): %s", tids.size(), frame_budget.count() / 1e6,
|
|
ok ? "on" : "unavailable");
|
|
}
|
|
const auto start = steady::now();
|
|
double update_ms = 0.0;
|
|
int completed = 0;
|
|
std::vector<double> frame_ms;
|
|
if (frames > 0) frame_ms.reserve(static_cast<std::size_t>(frames));
|
|
const auto cap_period = g_fps_cap > 0 ? std::chrono::nanoseconds(1'000'000'000LL / g_fps_cap)
|
|
: std::chrono::nanoseconds(0);
|
|
auto next_frame_at = steady::now();
|
|
while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) {
|
|
double pace_ms = 0.0;
|
|
if (cap_period.count()) {
|
|
// Fixed cadence: a late frame starts the next one at once, a frame more than one period
|
|
// late re-anchors instead of bursting to catch up.
|
|
const auto now = steady::now();
|
|
if (next_frame_at > now) {
|
|
std::this_thread::sleep_until(next_frame_at);
|
|
pace_ms = std::chrono::duration<double, std::milli>(steady::now() - now).count();
|
|
} else if (now - next_frame_at > cap_period) {
|
|
next_frame_at = now;
|
|
}
|
|
next_frame_at += cap_period;
|
|
}
|
|
const auto frame_start = steady::now();
|
|
const auto t0 = frame_start;
|
|
PythonBoot::UIUpdate();
|
|
const auto t_app = std::chrono::steady_clock::now();
|
|
renderer.sync_game_cursor();
|
|
PythonBoot::UIRender();
|
|
audio.pump();
|
|
const auto t1 = std::chrono::steady_clock::now();
|
|
update_ms += std::chrono::duration<double, std::milli>(t1 - t0).count();
|
|
unsigned ui_w = renderer.width(), ui_h = renderer.height();
|
|
UIRenderGetSize(&ui_w, &ui_h);
|
|
renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands());
|
|
const auto t2 = std::chrono::steady_clock::now();
|
|
const double elapsed_ms = std::chrono::duration<double, std::milli>(t2 - frame_start).count();
|
|
PythonBoot::SetNativeRenderFrameTime(static_cast<float>(elapsed_ms));
|
|
hint.report(std::int64_t((elapsed_ms - renderer.last_sync_ms()) * 1e6));
|
|
if (perf_log.enabled()) {
|
|
native_perf::FrameInput in;
|
|
in.frame_ms = elapsed_ms;
|
|
in.pace_ms = pace_ms;
|
|
in.app_ms = std::chrono::duration<double, std::milli>(t_app - t0).count();
|
|
in.host_ms = std::chrono::duration<double, std::milli>(t1 - t_app).count();
|
|
in.render_ms = std::chrono::duration<double, std::milli>(t2 - t1).count();
|
|
in.renderer = renderer.perf_totals();
|
|
perf_log.frame(in, [] { return PythonBoot::CurrentMapName(); });
|
|
}
|
|
frame_ms.push_back(elapsed_ms);
|
|
++completed;
|
|
}
|
|
renderer.finish_gpu_timings();
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: render loop complete (%d frames)", completed);
|
|
#endif
|
|
const auto wall_ms = std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now() - start).count();
|
|
print_summary(renderer, completed, wall_ms, update_ms, std::move(frame_ms));
|
|
|
|
PythonBoot::Stop();
|
|
#ifdef __ANDROID__
|
|
SDL_Log("live stage: PythonBoot stopped");
|
|
#endif
|
|
if (!use_external_server)
|
|
server.Stop();
|
|
return (frames == 0 || completed == frames) ? 0 : 2;
|
|
}
|
|
#endif
|
|
|
|
} // namespace
|
|
|
|
int main(int argc, char** argv) {
|
|
try {
|
|
int frames = 180;
|
|
bool frames_specified = false;
|
|
bool synthetic_specified = false;
|
|
bool interactive = false;
|
|
bool login_screen = false;
|
|
bool selection_screen = false;
|
|
int init_width = 1280;
|
|
int init_height = 800;
|
|
std::size_t draw_count = 64;
|
|
std::size_t triangles_per_draw = 333;
|
|
std::string capture_path;
|
|
std::string capture_out;
|
|
std::string screenshot_out;
|
|
std::string live_client_dir;
|
|
std::string live_server_spec;
|
|
int fake_mobs = 64;
|
|
bool animate_first_draw = false;
|
|
bool animate_bones = false;
|
|
bool vsync = true;
|
|
bool gpu_skinning = true;
|
|
bool native_terrain = true;
|
|
bool mobile_mode = false;
|
|
int safe_inset_px = 0;
|
|
bool show_fps = false;
|
|
int frames_in_flight = 2;
|
|
#ifdef __ANDROID__
|
|
mobile_mode = true;
|
|
#endif
|
|
for (int i = 1; i < argc; ++i) {
|
|
const std::string arg = argv[i];
|
|
if (arg == "--frames" && i+1 < argc) {
|
|
frames = std::stoi(argv[++i]);
|
|
frames_specified = true;
|
|
} else if (arg == "--interactive") interactive = true;
|
|
else if (arg == "--login-screen") {
|
|
login_screen = true;
|
|
interactive = true;
|
|
} else if (arg == "--selection-screen") {
|
|
selection_screen = true;
|
|
interactive = true;
|
|
} else if (arg == "--live-server" && i+1 < argc) {
|
|
live_server_spec = argv[++i];
|
|
interactive = true;
|
|
}
|
|
else if (arg == "--width" && i+1 < argc) init_width = std::stoi(argv[++i]);
|
|
else if (arg == "--height" && i+1 < argc) init_height = std::stoi(argv[++i]);
|
|
else if (arg == "--draws" && i+1 < argc) {
|
|
draw_count = std::stoul(argv[++i]);
|
|
synthetic_specified = true;
|
|
}
|
|
else if (arg == "--triangles-per-draw" && i+1 < argc) {
|
|
triangles_per_draw = std::stoul(argv[++i]);
|
|
synthetic_specified = true;
|
|
}
|
|
else if (arg == "--capture" && i+1 < argc) {
|
|
capture_path = argv[++i];
|
|
synthetic_specified = true;
|
|
}
|
|
else if (arg == "--capture-out" && i+1 < argc) capture_out = argv[++i];
|
|
else if (arg == "--screenshot-out" && i+1 < argc) screenshot_out = argv[++i];
|
|
else if (arg == "--live-client" && i+1 < argc) live_client_dir = argv[++i];
|
|
else if (arg == "--fake-mobs" && i+1 < argc) fake_mobs = std::stoi(argv[++i]);
|
|
else if (arg == "--animate-first-draw") {
|
|
animate_first_draw = true;
|
|
synthetic_specified = true;
|
|
}
|
|
else if (arg == "--animate-bones") {
|
|
animate_bones = true;
|
|
synthetic_specified = true;
|
|
}
|
|
else if (arg == "--no-vsync") vsync = false;
|
|
else if (arg == "--gpu-skinning") gpu_skinning = true;
|
|
else if (arg == "--no-gpu-skinning") gpu_skinning = false;
|
|
else if (arg == "--no-terrain") native_terrain = false;
|
|
else if (arg == "--mobile") mobile_mode = true;
|
|
else if (arg == "--safe-inset-px" && i+1 < argc) safe_inset_px = std::stoi(argv[++i]);
|
|
else if (arg == "--show-fps") show_fps = true;
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
else if (arg == "--py-exec" && i+1 < argc) g_py_exec_after_login = argv[++i];
|
|
else if (arg == "--perf-log") g_perf_log = true;
|
|
else if (arg == "--fps-cap" && i+1 < argc) g_fps_cap = std::max(0, std::stoi(argv[++i]));
|
|
else if (arg == "--refresh-rate" && i+1 < argc) g_refresh_rate = std::max(0, std::stoi(argv[++i]));
|
|
#endif
|
|
else if (arg == "--fake-idle-ms" && i+1 < argc) setenv("MT_FAKE_IDLE_MS", argv[++i], 1);
|
|
else if (arg == "--frames-in-flight" && i+1 < argc) frames_in_flight = std::stoi(argv[++i]);
|
|
else if (arg == "--msaa" && i+1 < argc) setenv("MT_MSAA", argv[++i], 1);
|
|
else if (arg == "--cpu-shadow") setenv("MT_CPU_SHADOW", "1", 1);
|
|
else throw std::runtime_error(
|
|
"usage: mt_native_render [--frames N] [--interactive] [--login-screen|--selection-screen] "
|
|
"[--live-server HOST:AUTH_PORT:GAME_PORT] [--width W] [--height H] "
|
|
"[--draws N] [--triangles-per-draw N] [--capture FILE] [--animate-first-draw] "
|
|
"[--animate-bones] [--no-vsync] [--live-client DIR] [--fake-mobs N] "
|
|
"[--gpu-skinning|--no-gpu-skinning] [--no-terrain] [--mobile] [--capture-out FILE] [--screenshot-out FILE.bmp] [--show-fps] [--perf-log] "
|
|
"[--fps-cap N] [--refresh-rate HZ] [--frames-in-flight 1|2] [--msaa 1|2|4|8] [--fake-idle-ms MS] [--cpu-shadow]");
|
|
}
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
if (native_perf::PerfLog::requested_by_env()) g_perf_log = true;
|
|
#endif
|
|
if (live_client_dir.empty() && !synthetic_specified) {
|
|
const char* env_client = std::getenv("MT_40250_CLIENT");
|
|
if (env_client && *env_client && access((std::string(env_client) + "/pack/Index").c_str(), R_OK) == 0) {
|
|
live_client_dir = env_client;
|
|
} else if (access("Client/pack/Index", R_OK) == 0) {
|
|
live_client_dir = "Client";
|
|
} else if (access("../Client/pack/Index", R_OK) == 0) {
|
|
live_client_dir = "../Client";
|
|
} else if (access("../../Client/pack/Index", R_OK) == 0) {
|
|
live_client_dir = "../../Client";
|
|
}
|
|
#ifdef __ANDROID__
|
|
if (live_client_dir.empty()) {
|
|
char* pref = SDL_GetPrefPath("mtgodot", "native-render");
|
|
if (pref) {
|
|
const std::string candidate = std::string(pref) + "Client";
|
|
if (access((candidate + "/pack/Index").c_str(), R_OK) == 0)
|
|
live_client_dir = candidate;
|
|
SDL_free(pref);
|
|
}
|
|
}
|
|
#endif
|
|
if (!live_client_dir.empty() && !frames_specified) {
|
|
interactive = true;
|
|
}
|
|
}
|
|
if (interactive && !frames_specified) frames = 0;
|
|
#ifdef __ANDROID__
|
|
SDL_Log("native startup: argc=%d live_client=%s selection=%d frames=%d", argc,
|
|
live_client_dir.c_str(), selection_screen, frames);
|
|
#endif
|
|
if (login_screen && selection_screen)
|
|
throw std::runtime_error("--login-screen and --selection-screen are mutually exclusive");
|
|
if (frames < 0 || (!interactive && frames == 0) || draw_count == 0 || draw_count > 4096 ||
|
|
triangles_per_draw == 0 || triangles_per_draw > 10000)
|
|
throw std::runtime_error("invalid frame/draw/triangle count");
|
|
VulkanWindow renderer(vsync, init_width, init_height);
|
|
renderer.set_safe_inset_px(safe_inset_px);
|
|
renderer.set_show_fps(show_fps);
|
|
renderer.set_frames_in_flight(std::uint32_t(std::max(1, frames_in_flight)));
|
|
if (!screenshot_out.empty() && frames > 0) renderer.request_screenshot(screenshot_out, std::uint64_t(frames));
|
|
if (mobile_mode) {
|
|
renderer.touch_controller.set_enabled(true);
|
|
#ifdef __ANDROID__
|
|
renderer.touch_controller.set_haptic(android_haptic);
|
|
#endif
|
|
renderer.touch_controller.update_screen_size(init_width, init_height);
|
|
}
|
|
if (!live_client_dir.empty()) {
|
|
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
|
|
const int result = run_live_client(
|
|
renderer,
|
|
live_client_dir,
|
|
frames,
|
|
fake_mobs,
|
|
gpu_skinning,
|
|
native_terrain,
|
|
login_screen,
|
|
selection_screen,
|
|
live_server_spec,
|
|
capture_out,
|
|
screenshot_out);
|
|
#ifdef __ANDROID__
|
|
SDL_Log("native live client exit: %d", result);
|
|
#endif
|
|
return result;
|
|
#else
|
|
throw std::runtime_error("mt_native_render was built without port_platform (--live-client unavailable)");
|
|
#endif
|
|
}
|
|
native_draw_capture::Capture capture{};
|
|
if (capture_path.empty()) {
|
|
capture.draws = make_test_draws(draw_count, triangles_per_draw);
|
|
} else {
|
|
capture = native_draw_capture::read_capture(capture_path);
|
|
}
|
|
auto& draws = capture.draws;
|
|
if (animate_first_draw && (draws.empty() || draws.front().positions.empty()))
|
|
throw std::runtime_error("first draw has no geometry to animate");
|
|
const auto start = std::chrono::steady_clock::now();
|
|
int completed = 0;
|
|
std::vector<double> frame_ms;
|
|
if (frames > 0) frame_ms.reserve(static_cast<std::size_t>(frames));
|
|
while ((frames == 0 || completed < frames) && renderer.poll()) {
|
|
const auto frame_start = std::chrono::steady_clock::now();
|
|
if (animate_first_draw && completed > 0) {
|
|
draws.front().positions[0] += 0.0001f;
|
|
++draws.front().geometry_revision;
|
|
}
|
|
if (animate_bones && completed > 0) {
|
|
const float offset = std::sin(float(completed) * 0.1f) * 0.5f;
|
|
for (auto& d : draws) {
|
|
for (std::size_t b = 12; b < d.bone_matrices.size(); b += 16)
|
|
d.bone_matrices[b] += offset;
|
|
}
|
|
}
|
|
renderer.render(draws, capture.textures, capture.ui_width, capture.ui_height, capture.ui_commands);
|
|
frame_ms.push_back(std::chrono::duration<double, std::milli>(
|
|
std::chrono::steady_clock::now() - frame_start).count());
|
|
++completed;
|
|
}
|
|
renderer.finish_gpu_timings();
|
|
const auto ms = std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now() - start).count();
|
|
print_summary(renderer, completed, ms, 0.0, std::move(frame_ms));
|
|
return (frames == 0 || completed == frames) ? 0 : 2;
|
|
} catch (const std::exception& error) {
|
|
std::cerr << "native renderer: " << error.what() << '\n';
|
|
#ifdef __ANDROID__
|
|
SDL_LogError(SDL_LOG_CATEGORY_APPLICATION, "native renderer: %s", error.what());
|
|
#endif
|
|
return 1;
|
|
}
|
|
}
|