feat: complete native client rendering parity updates
This commit is contained in:
@@ -5,12 +5,17 @@
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <iterator>
|
||||
#include <mutex>
|
||||
#include <unordered_map>
|
||||
|
||||
namespace {
|
||||
std::mutex g_draws_mutex;
|
||||
std::vector<Render3DDraw> g_draws;
|
||||
int g_native_terrain_override = -1;
|
||||
|
||||
|
||||
struct VertexLayout {
|
||||
unsigned stride = 0;
|
||||
@@ -69,9 +74,8 @@ int format_bytes(D3DFORMAT format)
|
||||
}
|
||||
|
||||
// PORT: device-created resources live in CPU memory with D3D reference counting. Textures made by
|
||||
// CreateTexture (CTerrain's splat alpha maps and attribute marks, CSnowEnvironment's blur targets)
|
||||
// keep their mip levels as bytes; nothing samples them because terrain and offscreen passes are
|
||||
// drawn by the Godot side.
|
||||
// CreateTexture resources keep their mip levels as bytes. Terrain alpha maps and the character
|
||||
// shadow render target are sampled by the native renderer through MtCpuMemoryTexture.
|
||||
struct CpuTexture;
|
||||
|
||||
struct CpuSurface final : IDirect3DSurface8
|
||||
@@ -89,13 +93,32 @@ struct CpuSurface final : IDirect3DSurface8
|
||||
HRESULT UnlockRect() override { return S_OK; }
|
||||
};
|
||||
|
||||
struct CpuTexture;
|
||||
std::mutex g_cpu_tex_mutex;
|
||||
std::unordered_map<unsigned, CpuTexture*> g_cpu_textures;
|
||||
unsigned g_next_cpu_tex_id = 1;
|
||||
|
||||
struct CpuTexture final : IDirect3DTexture8
|
||||
{
|
||||
struct Level { D3DSURFACE_DESC desc; std::vector<uint8_t> bytes; };
|
||||
|
||||
unsigned id = 0;
|
||||
std::uint32_t revision = 1;
|
||||
ULONG refs = 1;
|
||||
std::vector<Level> levels;
|
||||
|
||||
CpuTexture()
|
||||
{
|
||||
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
|
||||
id = g_next_cpu_tex_id++;
|
||||
g_cpu_textures[id] = this;
|
||||
}
|
||||
~CpuTexture()
|
||||
{
|
||||
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
|
||||
g_cpu_textures.erase(id);
|
||||
}
|
||||
|
||||
ULONG AddRef() override { return ++refs; }
|
||||
ULONG Release() override
|
||||
{
|
||||
@@ -132,7 +155,14 @@ struct CpuTexture final : IDirect3DTexture8
|
||||
locked->pBits = levels[level].bytes.data();
|
||||
return S_OK;
|
||||
}
|
||||
HRESULT UnlockRect(UINT level) override { return level < levels.size() ? S_OK : E_FAIL; }
|
||||
HRESULT UnlockRect(UINT level) override
|
||||
{
|
||||
if (level >= levels.size())
|
||||
return E_FAIL;
|
||||
if (level == 0)
|
||||
++revision;
|
||||
return S_OK;
|
||||
}
|
||||
};
|
||||
|
||||
ULONG CpuSurface::Release()
|
||||
@@ -217,6 +247,22 @@ public:
|
||||
m_renderTarget->AddRef();
|
||||
m_depthStencil = m_depthBuffer;
|
||||
m_depthStencil->AddRef();
|
||||
m_viewport = { 0, 0, (DWORD)width, (DWORD)height, 0.0f, 1.0f };
|
||||
m_renderStates[D3DRS_TEXTUREFACTOR] = 0xFFFFFFFF;
|
||||
m_renderStates[D3DRS_ZENABLE] = TRUE;
|
||||
m_renderStates[D3DRS_ZWRITEENABLE] = TRUE;
|
||||
m_renderStates[D3DRS_CULLMODE] = D3DCULL_CCW;
|
||||
m_renderStates[D3DRS_LIGHTING] = TRUE;
|
||||
m_renderStates[D3DRS_SRCBLEND] = D3DBLEND_ONE;
|
||||
m_renderStates[D3DRS_DESTBLEND] = D3DBLEND_ZERO;
|
||||
m_renderStates[D3DRS_ALPHAFUNC] = D3DCMP_ALWAYS;
|
||||
m_stageStates[0][D3DTSS_COLOROP] = D3DTOP_MODULATE;
|
||||
m_stageStates[0][D3DTSS_COLORARG1] = D3DTA_TEXTURE;
|
||||
m_stageStates[0][D3DTSS_COLORARG2] = D3DTA_CURRENT;
|
||||
m_stageStates[0][D3DTSS_ALPHAOP] = D3DTOP_SELECTARG1;
|
||||
m_stageStates[0][D3DTSS_ALPHAARG1] = D3DTA_TEXTURE;
|
||||
m_stageStates[1][D3DTSS_COLOROP] = D3DTOP_DISABLE;
|
||||
m_stageStates[1][D3DTSS_ALPHAOP] = D3DTOP_DISABLE;
|
||||
}
|
||||
~RecordingDevice() override
|
||||
{
|
||||
@@ -254,7 +300,14 @@ public:
|
||||
}
|
||||
HRESULT GetViewport(D3DVIEWPORT8* viewport) override
|
||||
{
|
||||
*viewport = { 0, 0, DWORD(m_width), DWORD(m_height), 0.0f, 1.0f };
|
||||
if (viewport)
|
||||
*viewport = m_viewport;
|
||||
return S_OK;
|
||||
}
|
||||
HRESULT SetViewport(const D3DVIEWPORT8* viewport) override
|
||||
{
|
||||
if (viewport)
|
||||
m_viewport = *viewport;
|
||||
return S_OK;
|
||||
}
|
||||
// PORT: CPU memory has no texture budget; report the 64MB of a mid-range 2004 card.
|
||||
@@ -340,8 +393,8 @@ public:
|
||||
*out = m_depthStencil;
|
||||
return S_OK;
|
||||
}
|
||||
// PORT: an offscreen target (CSnowEnvironment's blur pass) swallows its draws; only the back
|
||||
// buffer's draws are recorded for the Godot renderer.
|
||||
// Character shadow targets are rasterized below; other offscreen targets still need a
|
||||
// separate effect-specific implementation (snow blur is disabled by default).
|
||||
HRESULT SetRenderTarget(IDirect3DSurface8* target, IDirect3DSurface8* depth) override
|
||||
{
|
||||
if (target)
|
||||
@@ -358,7 +411,28 @@ public:
|
||||
}
|
||||
return S_OK;
|
||||
}
|
||||
HRESULT Clear(DWORD, const D3DRECT*, DWORD, D3DCOLOR, float, DWORD) override { return S_OK; }
|
||||
HRESULT Clear(DWORD, const D3DRECT*, DWORD flags, D3DCOLOR color, float depth, DWORD) override
|
||||
{
|
||||
if (m_renderTarget == m_backBuffer)
|
||||
return S_OK;
|
||||
auto* surface = static_cast<CpuSurface*>(m_renderTarget);
|
||||
if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5)
|
||||
return S_OK;
|
||||
const size_t pixels = size_t(surface->desc.Width) * surface->desc.Height;
|
||||
if (flags & D3DCLEAR_TARGET)
|
||||
{
|
||||
const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 |
|
||||
(((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3));
|
||||
auto& bytes = surface->parent->levels[0].bytes;
|
||||
bytes.resize(pixels * 2);
|
||||
for (size_t i = 0; i < pixels; ++i)
|
||||
std::memcpy(bytes.data() + i * 2, &rgb565, 2);
|
||||
++surface->parent->revision;
|
||||
}
|
||||
if (flags & D3DCLEAR_ZBUFFER)
|
||||
m_offscreenDepth.assign(pixels, depth);
|
||||
return S_OK;
|
||||
}
|
||||
HRESULT SetRenderState(D3DRENDERSTATETYPE state, DWORD value) override
|
||||
{
|
||||
if (unsigned(state) < kRenderStates)
|
||||
@@ -406,7 +480,8 @@ public:
|
||||
std::vector<std::uint32_t> indices(count);
|
||||
for (UINT i = 0; i < count; ++i)
|
||||
indices[i] = startVertex + i;
|
||||
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices);
|
||||
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices,
|
||||
m_stream, nullptr, startVertex);
|
||||
return S_OK;
|
||||
}
|
||||
HRESULT DrawIndexedPrimitive(D3DPRIMITIVETYPE type, UINT, UINT, UINT startIndex, UINT primitiveCount) override
|
||||
@@ -417,7 +492,8 @@ public:
|
||||
if (!read_indices(m_indices->bytes.data(), m_indices->bytes.size(), m_indices->format, startIndex,
|
||||
index_count(type, primitiveCount), m_baseVertex, &indices))
|
||||
return E_FAIL;
|
||||
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices);
|
||||
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices,
|
||||
m_stream, m_indices, startIndex);
|
||||
return S_OK;
|
||||
}
|
||||
HRESULT DrawPrimitiveUP(D3DPRIMITIVETYPE type, UINT primitiveCount, const void* vertices, UINT stride) override
|
||||
@@ -488,6 +564,92 @@ private:
|
||||
transform_point(&a[r * 4], b, &out[r * 4]);
|
||||
}
|
||||
|
||||
// The 40250 shadow pass renders a flat grey silhouette into an R5G6B5 target.
|
||||
// Rasterizing that small target here keeps the D3D render-target sequence and lets
|
||||
// the normal memory-texture path upload the result for terrain/object projection.
|
||||
void rasterize_shadow(const Render3DDraw& draw)
|
||||
{
|
||||
auto* surface = static_cast<CpuSurface*>(m_renderTarget);
|
||||
if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5 || draw.lines)
|
||||
return;
|
||||
const int width = int(surface->desc.Width), height = int(surface->desc.Height);
|
||||
if (width <= 0 || height <= 0 || draw.indices.size() < 3)
|
||||
return;
|
||||
auto& bytes = surface->parent->levels[0].bytes;
|
||||
if (bytes.size() < size_t(width) * height * 2)
|
||||
return;
|
||||
if (m_offscreenDepth.size() != size_t(width) * height)
|
||||
m_offscreenDepth.assign(size_t(width) * height, 1.0f);
|
||||
float worldView[16], mvp[16];
|
||||
multiply(draw.world, draw.view, worldView);
|
||||
multiply(worldView, draw.proj, mvp);
|
||||
struct Point { float x, y, z; bool valid; };
|
||||
std::vector<Point> points(draw.positions.size() / 3);
|
||||
for (size_t i = 0; i < points.size(); ++i)
|
||||
{
|
||||
float p[4] = { draw.positions[i * 3], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f };
|
||||
if (!draw.bone_matrices.empty() && draw.bone_weights.size() >= (i + 1) * 4 &&
|
||||
draw.bone_indices.size() >= (i + 1) * 4)
|
||||
{
|
||||
float skinned[4] = {};
|
||||
float total = 0.0f;
|
||||
for (int b = 0; b < 4; ++b)
|
||||
{
|
||||
const float weight = draw.bone_weights[i * 4 + b];
|
||||
const size_t bone = draw.bone_indices[i * 4 + b];
|
||||
if (weight <= 0.0f || (bone + 1) * 16 > draw.bone_matrices.size()) continue;
|
||||
float transformed[4];
|
||||
transform_point(p, draw.bone_matrices.data() + bone * 16, transformed);
|
||||
for (int c = 0; c < 4; ++c) skinned[c] += transformed[c] * weight;
|
||||
total += weight;
|
||||
}
|
||||
if (total > 0.0f) { p[0] = skinned[0]; p[1] = skinned[1]; p[2] = skinned[2]; }
|
||||
}
|
||||
float clip[4];
|
||||
transform_point(p, mvp, clip);
|
||||
const bool valid = clip[3] > 0.00001f;
|
||||
const float invW = valid ? 1.0f / clip[3] : 0.0f;
|
||||
points[i] = { draw.viewport[0] + (clip[0] * invW * 0.5f + 0.5f) * draw.viewport[2],
|
||||
draw.viewport[1] + (0.5f - clip[1] * invW * 0.5f) * draw.viewport[3],
|
||||
clip[2] * invW, valid };
|
||||
}
|
||||
const std::uint32_t color = m_renderStates[D3DRS_TEXTUREFACTOR];
|
||||
const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 |
|
||||
(((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3));
|
||||
auto edge = [](const Point& a, const Point& b, float x, float y) {
|
||||
return (x - a.x) * (b.y - a.y) - (y - a.y) * (b.x - a.x);
|
||||
};
|
||||
for (size_t i = 0; i + 2 < draw.indices.size(); i += 3)
|
||||
{
|
||||
if (draw.indices[i] >= points.size() || draw.indices[i + 1] >= points.size() ||
|
||||
draw.indices[i + 2] >= points.size()) continue;
|
||||
const Point& a = points[draw.indices[i]];
|
||||
const Point& b = points[draw.indices[i + 1]];
|
||||
const Point& c = points[draw.indices[i + 2]];
|
||||
if (!a.valid || !b.valid || !c.valid) continue;
|
||||
const float area = edge(a, b, c.x, c.y);
|
||||
if (std::fabs(area) < 0.00001f) continue;
|
||||
const int x0 = std::max(0, int(std::floor(std::min({a.x, b.x, c.x}))));
|
||||
const int y0 = std::max(0, int(std::floor(std::min({a.y, b.y, c.y}))));
|
||||
const int x1 = std::min(width - 1, int(std::ceil(std::max({a.x, b.x, c.x}))));
|
||||
const int y1 = std::min(height - 1, int(std::ceil(std::max({a.y, b.y, c.y}))));
|
||||
for (int y = y0; y <= y1; ++y)
|
||||
for (int x = x0; x <= x1; ++x)
|
||||
{
|
||||
const float px = float(x) + 0.5f, py = float(y) + 0.5f;
|
||||
const float wa = edge(b, c, px, py) / area;
|
||||
const float wb = edge(c, a, px, py) / area;
|
||||
const float wc = 1.0f - wa - wb;
|
||||
if (wa < 0.0f || wb < 0.0f || wc < 0.0f) continue;
|
||||
const float z = wa * a.z + wb * b.z + wc * c.z;
|
||||
const size_t pixel = size_t(y) * width + x;
|
||||
if (z < 0.0f || z > 1.0f || z > m_offscreenDepth[pixel]) continue;
|
||||
m_offscreenDepth[pixel] = z;
|
||||
std::memcpy(bytes.data() + pixel * 2, &rgb565, 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Each two-triangle quad of a UI-space draw becomes an image command: its screen corners through
|
||||
// world * view * projection, its stage-0 texture (or D3DTA_TFACTOR colour when stage 0 selects it)
|
||||
// and, when stage 1 generates coordinates from the camera-space position through D3DTS_TEXTURE1
|
||||
@@ -585,9 +747,10 @@ private:
|
||||
}
|
||||
|
||||
void record(D3DPRIMITIVETYPE type, UINT primitives, const uint8_t* vertices, size_t vertexBytes, UINT stride,
|
||||
const std::vector<std::uint32_t>& indices)
|
||||
const std::vector<std::uint32_t>& indices, const MtCpuVertexBuffer* source_vertex = nullptr,
|
||||
const MtCpuIndexBuffer* source_index = nullptr, UINT source_first_index = 0)
|
||||
{
|
||||
if (type == D3DPT_POINTLIST || indices.empty() || !vertices || m_renderTarget != m_backBuffer)
|
||||
if (type == D3DPT_POINTLIST || indices.empty() || !vertices)
|
||||
return;
|
||||
VertexLayout layout = fvf_layout(m_fvf);
|
||||
if (!stride)
|
||||
@@ -602,7 +765,7 @@ private:
|
||||
|
||||
// An orthographic projection over untransformed vertices is a UI-space draw (CPythonMiniMap's
|
||||
// terrain tiles under CPythonGraphic::SetOrtho2D): it joins the UI command stream in order.
|
||||
if (!layout.rhw && m_transforms[D3DTS_PROJECTION][11] == 0.0f)
|
||||
if (m_renderTarget == m_backBuffer && !layout.rhw && m_transforms[D3DTS_PROJECTION][11] == 0.0f)
|
||||
{
|
||||
record_ui_quads(type, vertices, vertexBytes, stride, layout, indices);
|
||||
return;
|
||||
@@ -614,6 +777,10 @@ private:
|
||||
std::memcpy(draw.proj, m_transforms[D3DTS_PROJECTION], sizeof(draw.proj));
|
||||
draw.texture0 = UIRenderTextureNameFromHandle(m_textures[0]);
|
||||
draw.texture1 = UIRenderTextureNameFromHandle(m_textures[1]);
|
||||
draw.viewport[0] = float(m_viewport.X);
|
||||
draw.viewport[1] = float(m_viewport.Y);
|
||||
draw.viewport[2] = float(m_viewport.Width);
|
||||
draw.viewport[3] = float(m_viewport.Height);
|
||||
draw.pretransformed = layout.rhw;
|
||||
draw.lines = type == D3DPT_LINELIST || type == D3DPT_LINESTRIP;
|
||||
|
||||
@@ -627,47 +794,221 @@ private:
|
||||
if ((size_t(hi) + 1) * stride > vertexBytes)
|
||||
return;
|
||||
const size_t count = size_t(hi - lo) + 1;
|
||||
draw.positions.resize(count * 3);
|
||||
if (layout.rhw) draw.rhw.resize(count);
|
||||
if (layout.normal >= 0) draw.normals.resize(count * 3);
|
||||
if (layout.uv0 >= 0) draw.uv0.resize(count * 2);
|
||||
if (layout.uv1 >= 0) draw.uv1.resize(count * 2);
|
||||
if (layout.diffuse >= 0) draw.diffuse.resize(count);
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
const bool texgen_cam_pos0 = layout.uv0 < 0 && m_textures[0] != nullptr &&
|
||||
(m_stageStates[0][D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == D3DTSS_TCI_CAMERASPACEPOSITION &&
|
||||
m_stageStates[0][D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2;
|
||||
const bool texgen_cam_pos1 = layout.uv1 < 0 && m_textures[1] != nullptr &&
|
||||
(m_stageStates[1][D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == D3DTSS_TCI_CAMERASPACEPOSITION &&
|
||||
m_stageStates[1][D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2;
|
||||
const bool tex_xform0 = layout.uv0 >= 0 && m_textures[0] != nullptr &&
|
||||
(m_stageStates[0][D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == 0 &&
|
||||
m_stageStates[0][D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2;
|
||||
GpuSkinSubrangeView skin_view;
|
||||
const bool gpu_skinned = source_vertex != nullptr &&
|
||||
LookupGpuSkinSubrange(vertices, stride, lo, hi, &skin_view);
|
||||
if (source_vertex)
|
||||
{
|
||||
const uint8_t* v = vertices + (lo + i) * stride;
|
||||
std::memcpy(&draw.positions[i * 3], v, 12);
|
||||
if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4);
|
||||
if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12);
|
||||
if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8);
|
||||
if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8);
|
||||
if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4);
|
||||
// The original buffers outlive individual draw calls. Keep their identity separate from
|
||||
// Unlock revisions so a static surface can keep its Godot mesh across frames.
|
||||
auto mix = [](std::uint64_t hash, std::uint64_t value) {
|
||||
return (hash ^ value) * 1099511628211ull;
|
||||
};
|
||||
std::uint64_t key = 14695981039346656037ull;
|
||||
if (gpu_skinned)
|
||||
{
|
||||
key = mix(key, skin_view.source_mesh_key);
|
||||
key = mix(key, source_index ? source_index->id : 0);
|
||||
key = mix(key, source_first_index);
|
||||
key = mix(key, lo - skin_view.mesh_base_vertex);
|
||||
key = mix(key, hi - skin_view.mesh_base_vertex);
|
||||
key = mix(key, primitives);
|
||||
key = mix(key, type);
|
||||
key = mix(key, stride);
|
||||
key = mix(key, m_fvf);
|
||||
draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull;
|
||||
draw.geometry_revision = 1ull;
|
||||
}
|
||||
else
|
||||
{
|
||||
key = mix(key, source_vertex->id);
|
||||
key = mix(key, source_index ? source_index->id : 0);
|
||||
key = mix(key, source_first_index);
|
||||
key = mix(key, lo);
|
||||
key = mix(key, hi);
|
||||
key = mix(key, primitives);
|
||||
key = mix(key, type);
|
||||
key = mix(key, stride);
|
||||
key = mix(key, m_fvf);
|
||||
if (texgen_cam_pos0)
|
||||
key = mix(key, static_cast<std::uint64_t>(reinterpret_cast<std::uintptr_t>(m_textures[0])));
|
||||
if (texgen_cam_pos1)
|
||||
key = mix(key, static_cast<std::uint64_t>(reinterpret_cast<std::uintptr_t>(m_textures[1])));
|
||||
draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull;
|
||||
std::uint64_t revision = mix(14695981039346656037ull, source_vertex->revision);
|
||||
revision = mix(revision, source_index ? source_index->revision : 0);
|
||||
if (tex_xform0)
|
||||
{
|
||||
const float* m = m_transforms[D3DTS_TEXTURE0];
|
||||
for (int mi : { 0, 1, 4, 5, 8, 9, 12, 13 })
|
||||
{
|
||||
std::uint32_t bits = 0;
|
||||
std::memcpy(&bits, &m[mi], sizeof(bits));
|
||||
revision = mix(revision, bits);
|
||||
}
|
||||
}
|
||||
// Camera-space texture coordinates are baked into the captured geometry.
|
||||
// The terrain splat and character-shadow matrices can change while the
|
||||
// vertex/index buffers remain unchanged, so they are part of its revision.
|
||||
for (int stage = 0; stage < 2; ++stage)
|
||||
{
|
||||
if ((stage == 0 && !texgen_cam_pos0) || (stage == 1 && !texgen_cam_pos1))
|
||||
continue;
|
||||
float worldView[16], worldTexture[16];
|
||||
multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView);
|
||||
multiply(worldView, m_transforms[stage == 0 ? D3DTS_TEXTURE0 : D3DTS_TEXTURE1], worldTexture);
|
||||
for (int mi : { 0, 1, 4, 5, 8, 9, 12, 13 })
|
||||
{
|
||||
std::uint32_t bits = 0;
|
||||
std::memcpy(&bits, &worldTexture[mi], sizeof(bits));
|
||||
revision = mix(revision, bits);
|
||||
}
|
||||
}
|
||||
draw.geometry_revision = revision & 0x7FFFFFFFFFFFFFFFull;
|
||||
}
|
||||
}
|
||||
const std::uint64_t frame_id = UIRenderFrameId();
|
||||
auto cached = draw.geometry_key ? m_geometry_cache.find(draw.geometry_key) : m_geometry_cache.end();
|
||||
const bool reuse = cached != m_geometry_cache.end() && !cached->second.changing &&
|
||||
cached->second.revision == draw.geometry_revision;
|
||||
if (reuse)
|
||||
{
|
||||
cached->second.last_used = frame_id;
|
||||
const Render3DDraw& prior = cached->second.geometry;
|
||||
draw.positions = prior.positions;
|
||||
draw.rhw = prior.rhw;
|
||||
draw.normals = prior.normals;
|
||||
draw.uv0 = prior.uv0;
|
||||
draw.uv1 = prior.uv1;
|
||||
draw.diffuse = prior.diffuse;
|
||||
draw.indices = prior.indices;
|
||||
draw.bone_indices = prior.bone_indices;
|
||||
draw.bone_weights = prior.bone_weights;
|
||||
}
|
||||
else
|
||||
{
|
||||
draw.positions.resize(count * 3);
|
||||
if (layout.rhw) draw.rhw.resize(count);
|
||||
if (layout.normal >= 0) draw.normals.resize(count * 3);
|
||||
if (layout.uv0 >= 0) draw.uv0.resize(count * 2);
|
||||
if (layout.uv1 >= 0) draw.uv1.resize(count * 2);
|
||||
if (layout.diffuse >= 0) draw.diffuse.resize(count);
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
const uint8_t* v = vertices + (lo + i) * stride;
|
||||
std::memcpy(&draw.positions[i * 3], v, 12);
|
||||
if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4);
|
||||
if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12);
|
||||
if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8);
|
||||
if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8);
|
||||
if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4);
|
||||
}
|
||||
if (texgen_cam_pos0)
|
||||
{
|
||||
float worldView[16], worldTex0[16];
|
||||
multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView);
|
||||
multiply(worldView, m_transforms[D3DTS_TEXTURE0], worldTex0);
|
||||
draw.uv0.resize(count * 2);
|
||||
const bool clamp0_u = m_stageStates[0][D3DTSS_ADDRESSU] == D3DTADDRESS_CLAMP;
|
||||
const bool clamp0_v = m_stageStates[0][D3DTSS_ADDRESSV] == D3DTADDRESS_CLAMP;
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
const float pos[4] = { draw.positions[i * 3 + 0], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f };
|
||||
float tc[4];
|
||||
transform_point(pos, worldTex0, tc);
|
||||
draw.uv0[i * 2 + 0] = clamp0_u ? std::clamp(tc[0], 0.5f / 256.0f, 255.5f / 256.0f) : tc[0];
|
||||
draw.uv0[i * 2 + 1] = clamp0_v ? std::clamp(tc[1], 0.5f / 256.0f, 255.5f / 256.0f) : tc[1];
|
||||
}
|
||||
}
|
||||
else if (tex_xform0)
|
||||
{
|
||||
const float* m = m_transforms[D3DTS_TEXTURE0];
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
const float u = draw.uv0[i * 2 + 0];
|
||||
const float v = draw.uv0[i * 2 + 1];
|
||||
draw.uv0[i * 2 + 0] = u * m[0] + v * m[4] + m[8] + m[12];
|
||||
draw.uv0[i * 2 + 1] = u * m[1] + v * m[5] + m[9] + m[13];
|
||||
}
|
||||
}
|
||||
if (texgen_cam_pos1)
|
||||
{
|
||||
float worldView[16], worldTex1[16];
|
||||
multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView);
|
||||
multiply(worldView, m_transforms[D3DTS_TEXTURE1], worldTex1);
|
||||
draw.uv1.resize(count * 2);
|
||||
for (size_t i = 0; i < count; ++i)
|
||||
{
|
||||
const float pos[4] = { draw.positions[i * 3 + 0], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f };
|
||||
float tc[4];
|
||||
transform_point(pos, worldTex1, tc);
|
||||
draw.uv1[i * 2 + 0] = tc[0];
|
||||
draw.uv1[i * 2 + 1] = tc[1];
|
||||
}
|
||||
}
|
||||
if (gpu_skinned)
|
||||
{
|
||||
const std::size_t rel_lo = std::size_t(lo - skin_view.mesh_base_vertex);
|
||||
draw.bone_indices.resize(count * 4);
|
||||
draw.bone_weights.resize(count * 4);
|
||||
std::memcpy(draw.bone_indices.data(), skin_view.bone_indices + rel_lo * 4, count * 4 * sizeof(std::uint8_t));
|
||||
std::memcpy(draw.bone_weights.data(), skin_view.bone_weights + rel_lo * 4, count * 4 * sizeof(float));
|
||||
}
|
||||
|
||||
// Strips and fans become lists; D3D flips the winding of every odd strip triangle.
|
||||
switch (type)
|
||||
{
|
||||
case D3DPT_TRIANGLESTRIP:
|
||||
for (UINT i = 0; i < primitives; ++i)
|
||||
{
|
||||
const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo;
|
||||
if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c });
|
||||
else draw.indices.insert(draw.indices.end(), { a, b, c });
|
||||
}
|
||||
break;
|
||||
case D3DPT_TRIANGLEFAN:
|
||||
for (UINT i = 0; i < primitives; ++i)
|
||||
draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo });
|
||||
break;
|
||||
case D3DPT_LINESTRIP:
|
||||
for (UINT i = 0; i < primitives; ++i)
|
||||
draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo });
|
||||
break;
|
||||
default:
|
||||
for (std::uint32_t index : indices)
|
||||
draw.indices.push_back(index - lo);
|
||||
break;
|
||||
}
|
||||
if (cached != m_geometry_cache.end())
|
||||
{
|
||||
cached->second.last_used = frame_id;
|
||||
if (!cached->second.changing && cached->second.revision != draw.geometry_revision)
|
||||
{
|
||||
cached->second.changing = true;
|
||||
cached->second.geometry = Render3DDraw();
|
||||
}
|
||||
}
|
||||
else if (draw.geometry_key && cached == m_geometry_cache.end())
|
||||
m_geometry_cache.emplace(draw.geometry_key, GeometryCacheEntry{draw.geometry_revision, frame_id, false, draw});
|
||||
}
|
||||
if (gpu_skinned && skin_view.bone_matrices && skin_view.bone_count > 0)
|
||||
draw.bone_matrices.assign(skin_view.bone_matrices, skin_view.bone_matrices + std::size_t(skin_view.bone_count) * 16);
|
||||
if (frame_id >= m_cache_pruned_frame + 120)
|
||||
{
|
||||
m_cache_pruned_frame = frame_id;
|
||||
for (auto it = m_geometry_cache.begin(); it != m_geometry_cache.end();)
|
||||
it = frame_id - it->second.last_used > 120 ? m_geometry_cache.erase(it) : std::next(it);
|
||||
}
|
||||
|
||||
// Strips and fans become lists; D3D flips the winding of every odd strip triangle.
|
||||
switch (type)
|
||||
{
|
||||
case D3DPT_TRIANGLESTRIP:
|
||||
for (UINT i = 0; i < primitives; ++i)
|
||||
{
|
||||
const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo;
|
||||
if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c });
|
||||
else draw.indices.insert(draw.indices.end(), { a, b, c });
|
||||
}
|
||||
break;
|
||||
case D3DPT_TRIANGLEFAN:
|
||||
for (UINT i = 0; i < primitives; ++i)
|
||||
draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo });
|
||||
break;
|
||||
case D3DPT_LINESTRIP:
|
||||
for (UINT i = 0; i < primitives; ++i)
|
||||
draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo });
|
||||
break;
|
||||
default:
|
||||
for (std::uint32_t index : indices)
|
||||
draw.indices.push_back(index - lo);
|
||||
break;
|
||||
}
|
||||
|
||||
draw.alpha_blend = m_renderStates[D3DRS_ALPHABLENDENABLE];
|
||||
draw.src_blend = m_renderStates[D3DRS_SRCBLEND];
|
||||
@@ -682,7 +1023,17 @@ private:
|
||||
draw.lighting = m_renderStates[D3DRS_LIGHTING];
|
||||
draw.texture_factor = m_renderStates[D3DRS_TEXTUREFACTOR];
|
||||
draw.fog_enable = m_renderStates[D3DRS_FOGENABLE];
|
||||
draw.fog_color = m_renderStates[D3DRS_FOGCOLOR];
|
||||
draw.fog_vertex_mode = m_renderStates[D3DRS_FOGVERTEXMODE];
|
||||
draw.fog_table_mode = m_renderStates[D3DRS_FOGTABLEMODE];
|
||||
draw.fog_range_enable = m_renderStates[D3DRS_RANGEFOGENABLE];
|
||||
std::memcpy(&draw.fog_start, &m_renderStates[D3DRS_FOGSTART], sizeof(float));
|
||||
std::memcpy(&draw.fog_end, &m_renderStates[D3DRS_FOGEND], sizeof(float));
|
||||
std::memcpy(&draw.fog_density, &m_renderStates[D3DRS_FOGDENSITY], sizeof(float));
|
||||
draw.ambient = m_renderStates[D3DRS_AMBIENT];
|
||||
draw.diffuse_material_source = m_renderStates[D3DRS_DIFFUSEMATERIALSOURCE];
|
||||
draw.ambient_material_source = m_renderStates[D3DRS_AMBIENTMATERIALSOURCE];
|
||||
draw.color_vertex = m_renderStates[D3DRS_COLORVERTEX];
|
||||
for (int stage = 0; stage < 2; ++stage)
|
||||
{
|
||||
draw.color_op[stage] = m_stageStates[stage][D3DTSS_COLOROP];
|
||||
@@ -691,11 +1042,16 @@ private:
|
||||
draw.alpha_op[stage] = m_stageStates[stage][D3DTSS_ALPHAOP];
|
||||
draw.alpha_arg1[stage] = m_stageStates[stage][D3DTSS_ALPHAARG1];
|
||||
draw.alpha_arg2[stage] = m_stageStates[stage][D3DTSS_ALPHAARG2];
|
||||
draw.address_u[stage] = m_stageStates[stage][D3DTSS_ADDRESSU];
|
||||
draw.address_v[stage] = m_stageStates[stage][D3DTSS_ADDRESSV];
|
||||
draw.min_filter[stage] = m_stageStates[stage][D3DTSS_MINFILTER];
|
||||
draw.mag_filter[stage] = m_stageStates[stage][D3DTSS_MAGFILTER];
|
||||
draw.mip_filter[stage] = m_stageStates[stage][D3DTSS_MIPFILTER];
|
||||
}
|
||||
copy_color(draw.material_diffuse, m_material.Diffuse);
|
||||
copy_color(draw.material_ambient, m_material.Ambient);
|
||||
copy_color(draw.material_emissive, m_material.Emissive);
|
||||
if (m_lightEnabled[0] && m_lights[0].Type == D3DLIGHT_DIRECTIONAL)
|
||||
if (m_lightEnabled[0] && (m_lights[0].Type == D3DLIGHT_DIRECTIONAL || m_lights[0].Type == D3DLIGHT_SPOT))
|
||||
{
|
||||
draw.light0 = true;
|
||||
draw.light0_direction[0] = m_lights[0].Direction.x;
|
||||
@@ -704,11 +1060,37 @@ private:
|
||||
copy_color(draw.light0_diffuse, m_lights[0].Diffuse);
|
||||
copy_color(draw.light0_ambient, m_lights[0].Ambient);
|
||||
}
|
||||
Render3DAdd(std::move(draw));
|
||||
for (int i = 0; i < 2; ++i)
|
||||
{
|
||||
if (!m_lightEnabled[i]) continue;
|
||||
const D3DLIGHT8& source = m_lights[i];
|
||||
auto& light = draw.lights[i];
|
||||
light.type = source.Type;
|
||||
light.position[0] = source.Position.x;
|
||||
light.position[1] = source.Position.y;
|
||||
light.position[2] = source.Position.z;
|
||||
light.direction[0] = source.Direction.x;
|
||||
light.direction[1] = source.Direction.y;
|
||||
light.direction[2] = source.Direction.z;
|
||||
copy_color(light.diffuse, source.Diffuse);
|
||||
copy_color(light.ambient, source.Ambient);
|
||||
light.attenuation[0] = source.Attenuation0;
|
||||
light.attenuation[1] = source.Attenuation1;
|
||||
light.attenuation[2] = source.Attenuation2;
|
||||
light.range = source.Range;
|
||||
light.theta = source.Theta;
|
||||
light.phi = source.Phi;
|
||||
light.falloff = source.Falloff;
|
||||
}
|
||||
if (m_renderTarget == m_backBuffer)
|
||||
Render3DAdd(std::move(draw));
|
||||
else
|
||||
rasterize_shadow(draw);
|
||||
}
|
||||
|
||||
ULONG m_refs = 1;
|
||||
int m_width, m_height;
|
||||
D3DVIEWPORT8 m_viewport = {};
|
||||
float m_transforms[kTransforms][16];
|
||||
DWORD m_renderStates[kRenderStates] = {};
|
||||
DWORD m_stageStates[kStages][kStageStates] = {};
|
||||
@@ -725,11 +1107,103 @@ private:
|
||||
UINT m_streamStride = 0;
|
||||
MtCpuIndexBuffer* m_indices = nullptr;
|
||||
UINT m_baseVertex = 0;
|
||||
struct GeometryCacheEntry {
|
||||
std::uint64_t revision;
|
||||
std::uint64_t last_used;
|
||||
bool changing;
|
||||
Render3DDraw geometry;
|
||||
};
|
||||
std::unordered_map<std::uint64_t, GeometryCacheEntry> m_geometry_cache;
|
||||
std::uint64_t m_cache_pruned_frame = 0;
|
||||
std::vector<float> m_offscreenDepth;
|
||||
};
|
||||
}
|
||||
|
||||
std::string MtCpuTextureNameFromHandle(const IDirect3DBaseTexture8* handle)
|
||||
{
|
||||
if (!handle) return {};
|
||||
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
|
||||
for (const auto& entry : g_cpu_textures)
|
||||
{
|
||||
if (entry.second == handle && !entry.second->levels.empty())
|
||||
return "mem:cpu_" + std::to_string(entry.first) + "@" + std::to_string(entry.second->revision);
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
bool MtCpuMemoryTexture(const std::string& name, UIMemoryTexture* out)
|
||||
{
|
||||
if (name.compare(0, 8, "mem:cpu_") != 0 || !out) return false;
|
||||
const unsigned id = unsigned(std::strtoul(name.c_str() + 8, nullptr, 10));
|
||||
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
|
||||
auto it = g_cpu_textures.find(id);
|
||||
if (it == g_cpu_textures.end() || it->second->levels.empty()) return false;
|
||||
const auto& lv0 = it->second->levels[0];
|
||||
const UINT w = lv0.desc.Width;
|
||||
const UINT h = lv0.desc.Height;
|
||||
if (!w || !h) return false;
|
||||
out->width = static_cast<int>(w);
|
||||
out->height = static_cast<int>(h);
|
||||
out->revision = it->second->revision;
|
||||
out->argb.resize(std::size_t(w) * std::size_t(h));
|
||||
const uint8_t* bytes = lv0.bytes.data();
|
||||
const int bpp = format_bytes(lv0.desc.Format);
|
||||
for (std::size_t i = 0; i < out->argb.size(); ++i)
|
||||
{
|
||||
if (bpp == 4)
|
||||
{
|
||||
std::uint32_t px = 0;
|
||||
std::memcpy(&px, bytes + i * 4, 4);
|
||||
// CTerrain::PutImage32 packs alpha into bits 31..24 with RGB == 0.
|
||||
// Set RGB to white so stage-1 modulation preserves the base texture's RGB.
|
||||
if ((px & 0x00FFFFFFu) == 0)
|
||||
px |= 0x00FFFFFFu;
|
||||
out->argb[i] = px;
|
||||
}
|
||||
else if (lv0.desc.Format == D3DFMT_R5G6B5)
|
||||
{
|
||||
std::uint16_t word = 0;
|
||||
std::memcpy(&word, bytes + i * 2, 2);
|
||||
const std::uint32_t r = ((word >> 11) & 31u) * 255u / 31u;
|
||||
const std::uint32_t g = ((word >> 5) & 63u) * 255u / 63u;
|
||||
const std::uint32_t b = (word & 31u) * 255u / 31u;
|
||||
out->argb[i] = 0xFF000000u | (r << 16) | (g << 8) | b;
|
||||
}
|
||||
else if (bpp == 2)
|
||||
{
|
||||
std::uint16_t word = 0;
|
||||
std::memcpy(&word, bytes + i * 2, 2);
|
||||
// CTerrain::PutImage16 writes `src[x] << 8`, storing the full 8-bit alpha in the high byte.
|
||||
const std::uint32_t a = (word >> 8) & 0xFFu;
|
||||
out->argb[i] = (a << 24) | 0x00FFFFFFu;
|
||||
}
|
||||
else
|
||||
{
|
||||
const std::uint32_t a = bytes[i];
|
||||
out->argb[i] = (a << 24) | 0x00FFFFFFu;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
IDirect3DDevice8* MtCreateRecordingDevice(int width, int height) { return new RecordingDevice(width, height); }
|
||||
|
||||
void SetNativeTerrainRenderEnabled(bool enabled)
|
||||
{
|
||||
g_native_terrain_override = enabled ? 1 : 0;
|
||||
}
|
||||
|
||||
bool IsNativeTerrainRenderEnabled()
|
||||
{
|
||||
if (g_native_terrain_override >= 0)
|
||||
return g_native_terrain_override != 0;
|
||||
static const bool env_on = [] {
|
||||
const char* v = std::getenv("MT_NATIVE_TERRAIN");
|
||||
return v && (*v == '1' || *v == 't' || *v == 'T' || *v == 'y' || *v == 'Y');
|
||||
}();
|
||||
return env_on;
|
||||
}
|
||||
|
||||
void Render3DBeginFrame()
|
||||
{
|
||||
std::lock_guard<std::mutex> lock(g_draws_mutex);
|
||||
|
||||
Reference in New Issue
Block a user