Files
mtgodot-poc/extension/src/platform/EterLib/RecordingDevice.cpp
T

1260 lines
46 KiB
C++

#include "RecordingDevice.h"
#include "RenderCommands3D.h"
#include "UIRenderCommands.h"
#include "CpuBuffer.h"
#include <algorithm>
#include <cmath>
#include <cstdlib>
#include <cstring>
#include <iterator>
#include <mutex>
#include <unordered_map>
namespace {
std::mutex g_draws_mutex;
std::vector<Render3DDraw> g_draws;
int g_native_terrain_override = -1;
struct VertexLayout {
unsigned stride = 0;
bool rhw = false;
int normal = -1, diffuse = -1, uv0 = -1, uv1 = -1;
};
// The element offsets of a fixed-function FVF (D3DXGetFVFVertexSize's order).
VertexLayout fvf_layout(DWORD fvf)
{
VertexLayout layout;
unsigned offset = 0;
switch (fvf & D3DFVF_POSITION_MASK)
{
case D3DFVF_XYZ: offset = 12; break;
case D3DFVF_XYZRHW: offset = 16; layout.rhw = true; break;
case D3DFVF_XYZB1: offset = 16; break;
case D3DFVF_XYZB2: offset = 20; break;
case D3DFVF_XYZB3: offset = 24; break;
case D3DFVF_XYZB4: offset = 28; break;
case D3DFVF_XYZB5: offset = 32; break;
}
if (fvf & D3DFVF_NORMAL) { layout.normal = offset; offset += 12; }
if (fvf & D3DFVF_PSIZE) offset += 4;
if (fvf & D3DFVF_DIFFUSE) { layout.diffuse = offset; offset += 4; }
if (fvf & D3DFVF_SPECULAR) offset += 4;
const unsigned texCount = (fvf & D3DFVF_TEXCOUNT_MASK) >> D3DFVF_TEXCOUNT_SHIFT;
for (unsigned i = 0; i < texCount; ++i)
{
static const unsigned coordSize[4] = { 8, 12, 16, 4 };
if (i == 0) layout.uv0 = offset;
if (i == 1) layout.uv1 = offset;
offset += coordSize[(fvf >> (16 + i * 2)) & 3];
}
layout.stride = offset;
return layout;
}
void copy_color(float out[4], const D3DCOLORVALUE& c)
{
out[0] = c.r; out[1] = c.g; out[2] = c.b; out[3] = c.a;
}
int format_bytes(D3DFORMAT format)
{
switch (format)
{
case D3DFMT_A4R4G4B4: case D3DFMT_R5G6B5: case D3DFMT_A1R5G5B5: case D3DFMT_X1R5G5B5: case D3DFMT_X4R4G4B4:
case D3DFMT_D16: case D3DFMT_D16_LOCKABLE:
return 2;
case D3DFMT_A8: case D3DFMT_L8:
return 1;
default:
return 4;
}
}
// PORT: device-created resources live in CPU memory with D3D reference counting. Textures made by
// CreateTexture resources keep their mip levels as bytes. Terrain alpha maps and the character
// shadow render target are sampled by the native renderer through MtCpuMemoryTexture.
struct CpuTexture;
struct CpuSurface final : IDirect3DSurface8
{
ULONG refs = 1;
CpuTexture* parent = nullptr; // a texture level, or null for a standalone surface
UINT level = 0;
D3DSURFACE_DESC desc = {};
std::vector<uint8_t> bytes; // standalone surfaces only
ULONG AddRef() override { return ++refs; }
ULONG Release() override;
HRESULT GetDesc(D3DSURFACE_DESC* out) override { *out = desc; return S_OK; }
HRESULT LockRect(D3DLOCKED_RECT* locked, const RECT*, DWORD) override;
HRESULT UnlockRect() override { return S_OK; }
};
struct CpuTexture;
std::mutex g_cpu_tex_mutex;
std::unordered_map<unsigned, CpuTexture*> g_cpu_textures;
unsigned g_next_cpu_tex_id = 1;
struct CpuTexture final : IDirect3DTexture8
{
struct Level { D3DSURFACE_DESC desc; std::vector<uint8_t> bytes; };
unsigned id = 0;
std::uint32_t revision = 1;
ULONG refs = 1;
std::vector<Level> levels;
CpuTexture()
{
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
id = g_next_cpu_tex_id++;
g_cpu_textures[id] = this;
}
~CpuTexture()
{
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
g_cpu_textures.erase(id);
}
ULONG AddRef() override { return ++refs; }
ULONG Release() override
{
const ULONG left = --refs;
if (!left)
delete this;
return left;
}
DWORD GetLevelCount() override { return DWORD(levels.size()); }
HRESULT GetLevelDesc(UINT level, D3DSURFACE_DESC* out) override
{
if (level >= levels.size())
return E_FAIL;
*out = levels[level].desc;
return S_OK;
}
HRESULT GetSurfaceLevel(UINT level, IDirect3DSurface8** out) override
{
if (level >= levels.size())
return E_FAIL;
auto* surface = new CpuSurface();
surface->parent = this;
surface->level = level;
surface->desc = levels[level].desc;
AddRef();
*out = surface;
return S_OK;
}
HRESULT LockRect(UINT level, D3DLOCKED_RECT* locked, const RECT*, DWORD) override
{
if (level >= levels.size())
return E_FAIL;
locked->Pitch = INT(levels[level].desc.Width * format_bytes(levels[level].desc.Format));
locked->pBits = levels[level].bytes.data();
return S_OK;
}
HRESULT UnlockRect(UINT level) override
{
if (level >= levels.size())
return E_FAIL;
if (level == 0)
++revision;
return S_OK;
}
};
ULONG CpuSurface::Release()
{
const ULONG left = --refs;
if (!left)
{
if (parent)
parent->Release();
delete this;
}
return left;
}
HRESULT CpuSurface::LockRect(D3DLOCKED_RECT* locked, const RECT* rect, DWORD flags)
{
if (parent)
return parent->LockRect(level, locked, rect, flags);
locked->Pitch = INT(desc.Width * format_bytes(desc.Format));
locked->pBits = bytes.data();
return S_OK;
}
D3DSURFACE_DESC surface_desc(UINT width, UINT height, D3DFORMAT format, DWORD usage, D3DPOOL pool)
{
D3DSURFACE_DESC desc = {};
desc.Format = format;
desc.Type = D3DRTYPE_SURFACE;
desc.Usage = usage;
desc.Pool = pool;
desc.Size = width * height * format_bytes(format);
desc.MultiSampleType = D3DMULTISAMPLE_NONE;
desc.Width = width;
desc.Height = height;
return desc;
}
struct DeviceVertexBuffer final : MtCpuVertexBuffer
{
ULONG refs = 1;
ULONG AddRef() override { return ++refs; }
ULONG Release() override
{
const ULONG left = --refs;
if (!left)
delete this;
return left;
}
};
struct DeviceIndexBuffer final : MtCpuIndexBuffer
{
ULONG refs = 1;
ULONG AddRef() override { return ++refs; }
ULONG Release() override
{
const ULONG left = --refs;
if (!left)
delete this;
return left;
}
};
class RecordingDevice final : public IDirect3DDevice8
{
public:
RecordingDevice(int width, int height) : m_width(width), m_height(height)
{
static const float identity[16] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 };
for (auto& matrix : m_transforms)
std::memcpy(matrix, identity, sizeof(identity));
std::memset(&m_material, 0, sizeof(m_material));
m_material.Diffuse = { 1, 1, 1, 1 };
std::memset(m_lights, 0, sizeof(m_lights));
auto* backBuffer = new CpuSurface();
backBuffer->desc = surface_desc(width, height, D3DFMT_X8R8G8B8, D3DUSAGE_RENDERTARGET, D3DPOOL_DEFAULT);
m_backBuffer = backBuffer;
auto* depthBuffer = new CpuSurface();
depthBuffer->desc = surface_desc(width, height, D3DFMT_D16, D3DUSAGE_DEPTHSTENCIL, D3DPOOL_DEFAULT);
m_depthBuffer = depthBuffer;
m_renderTarget = m_backBuffer;
m_renderTarget->AddRef();
m_depthStencil = m_depthBuffer;
m_depthStencil->AddRef();
m_viewport = { 0, 0, (DWORD)width, (DWORD)height, 0.0f, 1.0f };
m_renderStates[D3DRS_TEXTUREFACTOR] = 0xFFFFFFFF;
m_renderStates[D3DRS_ZENABLE] = TRUE;
m_renderStates[D3DRS_ZWRITEENABLE] = TRUE;
m_renderStates[D3DRS_CULLMODE] = D3DCULL_CCW;
m_renderStates[D3DRS_LIGHTING] = TRUE;
m_renderStates[D3DRS_SRCBLEND] = D3DBLEND_ONE;
m_renderStates[D3DRS_DESTBLEND] = D3DBLEND_ZERO;
m_renderStates[D3DRS_ALPHAFUNC] = D3DCMP_ALWAYS;
m_stageStates[0][D3DTSS_COLOROP] = D3DTOP_MODULATE;
m_stageStates[0][D3DTSS_COLORARG1] = D3DTA_TEXTURE;
m_stageStates[0][D3DTSS_COLORARG2] = D3DTA_CURRENT;
m_stageStates[0][D3DTSS_ALPHAOP] = D3DTOP_SELECTARG1;
m_stageStates[0][D3DTSS_ALPHAARG1] = D3DTA_TEXTURE;
m_stageStates[1][D3DTSS_COLOROP] = D3DTOP_DISABLE;
m_stageStates[1][D3DTSS_ALPHAOP] = D3DTOP_DISABLE;
}
~RecordingDevice() override
{
m_renderTarget->Release();
m_depthStencil->Release();
m_backBuffer->Release();
m_depthBuffer->Release();
}
ULONG AddRef() override { return ++m_refs; }
ULONG Release() override
{
const ULONG refs = --m_refs;
if (!refs)
delete this;
return refs;
}
// PORT: a D3D8 HAL on a T&L card (D3DDEVCAPS_HWTRANSFORMANDLIGHT), 8 lights, 4 texture stages.
HRESULT GetDeviceCaps(D3DCAPS8* caps) override
{
std::memset(caps, 0, sizeof(*caps));
caps->DevCaps = D3DDEVCAPS_HWTRANSFORMANDLIGHT;
caps->MaxActiveLights = 8;
caps->MaxTextureBlendStages = 4;
caps->MaxSimultaneousTextures = 4;
caps->MaxTextureWidth = caps->MaxTextureHeight = 4096;
caps->MaxAnisotropy = 16;
caps->MaxPrimitiveCount = 0xFFFFF;
caps->MaxVertexIndex = 0xFFFFF;
caps->MaxStreams = 8;
caps->MaxStreamStride = 255;
caps->TextureAddressCaps = D3DPTADDRESSCAPS_BORDER | D3DPTADDRESSCAPS_CLAMP | D3DPTADDRESSCAPS_WRAP | D3DPTADDRESSCAPS_MIRROR;
return S_OK;
}
HRESULT GetViewport(D3DVIEWPORT8* viewport) override
{
if (viewport)
*viewport = m_viewport;
return S_OK;
}
HRESULT SetViewport(const D3DVIEWPORT8* viewport) override
{
if (viewport)
m_viewport = *viewport;
return S_OK;
}
// PORT: CPU memory has no texture budget; report the 64MB of a mid-range 2004 card.
UINT GetAvailableTextureMem() override { return 64u << 20; }
HRESULT BeginScene() override { return S_OK; }
HRESULT EndScene() override { return S_OK; }
HRESULT SetTransform(D3DTRANSFORMSTATETYPE state, const D3DMATRIX* matrix) override
{
if (unsigned(state) < kTransforms && matrix)
std::memcpy(m_transforms[state], matrix, sizeof(float) * 16);
return S_OK;
}
HRESULT GetTransform(D3DTRANSFORMSTATETYPE state, D3DMATRIX* matrix) override
{
if (unsigned(state) < kTransforms)
std::memcpy(matrix, m_transforms[state], sizeof(float) * 16);
return S_OK;
}
HRESULT SetMaterial(const D3DMATERIAL8* material) override { m_material = *material; return S_OK; }
HRESULT SetLight(DWORD index, const D3DLIGHT8* light) override
{
if (index < 8)
m_lights[index] = *light;
return S_OK;
}
HRESULT LightEnable(DWORD index, BOOL enable) override
{
if (index < 8)
m_lightEnabled[index] = enable != FALSE;
return S_OK;
}
HRESULT CreateTexture(UINT width, UINT height, UINT levels, DWORD usage, D3DFORMAT format, D3DPOOL pool, IDirect3DTexture8** out) override
{
if (!width || !height)
return E_FAIL;
auto* texture = new CpuTexture();
// Levels == 0 builds the full chain down to 1x1.
for (UINT w = width, h = height; ; w = std::max(1u, w / 2), h = std::max(1u, h / 2))
{
CpuTexture::Level level;
level.desc = surface_desc(w, h, format, usage, pool);
level.bytes.resize(level.desc.Size);
texture->levels.push_back(std::move(level));
if ((levels && texture->levels.size() == levels) || (w == 1 && h == 1))
break;
}
*out = texture;
return S_OK;
}
HRESULT CreateVertexBuffer(UINT length, DWORD, DWORD fvf, D3DPOOL, IDirect3DVertexBuffer8** out) override
{
auto* buffer = new DeviceVertexBuffer();
buffer->bytes.resize(length);
buffer->fvf = fvf;
*out = buffer;
return S_OK;
}
HRESULT CreateIndexBuffer(UINT length, DWORD, D3DFORMAT format, D3DPOOL, IDirect3DIndexBuffer8** out) override
{
auto* buffer = new DeviceIndexBuffer();
buffer->bytes.resize(length);
buffer->format = format;
*out = buffer;
return S_OK;
}
HRESULT CreateDepthStencilSurface(UINT width, UINT height, D3DFORMAT format, D3DMULTISAMPLE_TYPE, IDirect3DSurface8** out) override
{
auto* surface = new CpuSurface();
surface->desc = surface_desc(width, height, format, D3DUSAGE_DEPTHSTENCIL, D3DPOOL_DEFAULT);
*out = surface;
return S_OK;
}
HRESULT GetRenderTarget(IDirect3DSurface8** out) override
{
m_renderTarget->AddRef();
*out = m_renderTarget;
return S_OK;
}
HRESULT GetDepthStencilSurface(IDirect3DSurface8** out) override
{
m_depthStencil->AddRef();
*out = m_depthStencil;
return S_OK;
}
// Character shadow targets are rasterized below; other offscreen targets still need a
// separate effect-specific implementation (snow blur is disabled by default).
HRESULT SetRenderTarget(IDirect3DSurface8* target, IDirect3DSurface8* depth) override
{
if (target)
{
target->AddRef();
m_renderTarget->Release();
m_renderTarget = target;
}
if (depth)
{
depth->AddRef();
m_depthStencil->Release();
m_depthStencil = depth;
}
return S_OK;
}
HRESULT Clear(DWORD, const D3DRECT*, DWORD flags, D3DCOLOR color, float depth, DWORD) override
{
if (m_renderTarget == m_backBuffer)
{
// Kept in draw order: the renderer clears its colour/depth attachments at this point.
Render3DDraw clear;
clear.clear_flags = flags & (D3DCLEAR_TARGET | D3DCLEAR_ZBUFFER | D3DCLEAR_STENCIL);
clear.clear_color = color;
clear.clear_z = depth;
if (clear.clear_flags)
Render3DAdd(std::move(clear));
return S_OK;
}
auto* surface = static_cast<CpuSurface*>(m_renderTarget);
if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5)
return S_OK;
const size_t pixels = size_t(surface->desc.Width) * surface->desc.Height;
if (flags & D3DCLEAR_TARGET)
{
const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 |
(((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3));
auto& bytes = surface->parent->levels[0].bytes;
bytes.resize(pixels * 2);
for (size_t i = 0; i < pixels; ++i)
std::memcpy(bytes.data() + i * 2, &rgb565, 2);
++surface->parent->revision;
}
if (flags & D3DCLEAR_ZBUFFER)
m_offscreenDepth.assign(pixels, depth);
return S_OK;
}
HRESULT SetRenderState(D3DRENDERSTATETYPE state, DWORD value) override
{
if (unsigned(state) < kRenderStates)
m_renderStates[state] = value;
return S_OK;
}
HRESULT SetTexture(DWORD stage, IDirect3DBaseTexture8* texture) override
{
if (stage < kStages)
m_textures[stage] = texture;
return S_OK;
}
HRESULT SetTextureStageState(DWORD stage, D3DTEXTURESTAGESTATETYPE type, DWORD value) override
{
if (stage < kStages && unsigned(type) < kStageStates)
m_stageStates[stage][type] = value;
return S_OK;
}
// PORT: fixed function only; CGraphicDevice's stream "shaders" are the FVFs they describe.
HRESULT SetVertexShader(DWORD handle) override { m_fvf = handle; return S_OK; }
HRESULT SetPixelShader(DWORD) override { return S_OK; }
HRESULT SetVertexShaderConstant(DWORD, const void*, DWORD) override { return S_OK; }
HRESULT SetPixelShaderConstant(DWORD, const void*, DWORD) override { return S_OK; }
HRESULT SetStreamSource(UINT stream, IDirect3DVertexBuffer8* buffer, UINT stride) override
{
if (stream == 0)
{
m_stream = static_cast<MtCpuVertexBuffer*>(buffer);
m_streamStride = stride;
}
return S_OK;
}
HRESULT SetIndices(IDirect3DIndexBuffer8* indices, UINT baseVertex) override
{
m_indices = static_cast<MtCpuIndexBuffer*>(indices);
m_baseVertex = baseVertex;
return S_OK;
}
HRESULT DrawPrimitive(D3DPRIMITIVETYPE type, UINT startVertex, UINT primitiveCount) override
{
if (!m_stream)
return E_FAIL;
const UINT count = index_count(type, primitiveCount);
std::vector<std::uint32_t> indices(count);
for (UINT i = 0; i < count; ++i)
indices[i] = startVertex + i;
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices,
m_stream, nullptr, startVertex);
return S_OK;
}
HRESULT DrawIndexedPrimitive(D3DPRIMITIVETYPE type, UINT, UINT, UINT startIndex, UINT primitiveCount) override
{
if (!m_stream || !m_indices)
return E_FAIL;
std::vector<std::uint32_t> indices;
if (!read_indices(m_indices->bytes.data(), m_indices->bytes.size(), m_indices->format, startIndex,
index_count(type, primitiveCount), m_baseVertex, &indices))
return E_FAIL;
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices,
m_stream, m_indices, startIndex);
return S_OK;
}
HRESULT DrawPrimitiveUP(D3DPRIMITIVETYPE type, UINT primitiveCount, const void* vertices, UINT stride) override
{
const UINT count = index_count(type, primitiveCount);
std::vector<std::uint32_t> indices(count);
for (UINT i = 0; i < count; ++i)
indices[i] = i;
record(type, primitiveCount, static_cast<const uint8_t*>(vertices), size_t(count) * stride, stride, indices);
return S_OK;
}
HRESULT DrawIndexedPrimitiveUP(D3DPRIMITIVETYPE type, UINT minVertex, UINT numVertices, UINT primitiveCount,
const void* indexData, D3DFORMAT indexFormat, const void* vertices, UINT stride) override
{
const UINT count = index_count(type, primitiveCount);
const size_t indexSize = indexFormat == D3DFMT_INDEX32 ? 4 : 2;
std::vector<std::uint32_t> indices;
if (!read_indices(static_cast<const uint8_t*>(indexData), count * indexSize, indexFormat, 0, count, 0, &indices))
return E_FAIL;
record(type, primitiveCount, static_cast<const uint8_t*>(vertices), size_t(minVertex + numVertices) * stride, stride, indices);
return S_OK;
}
private:
static constexpr unsigned kTransforms = 512; // D3DTS_WORLDMATRIX(0..255) = 256..511
static constexpr unsigned kRenderStates = 256;
static constexpr unsigned kStages = 8;
static constexpr unsigned kStageStates = 32;
static UINT index_count(D3DPRIMITIVETYPE type, UINT primitives)
{
switch (type)
{
case D3DPT_POINTLIST: return primitives;
case D3DPT_LINELIST: return primitives * 2;
case D3DPT_LINESTRIP: return primitives + 1;
case D3DPT_TRIANGLELIST: return primitives * 3;
case D3DPT_TRIANGLESTRIP: case D3DPT_TRIANGLEFAN: return primitives + 2;
default: return 0;
}
}
static bool read_indices(const uint8_t* bytes, size_t size, D3DFORMAT format, UINT start, UINT count, UINT base,
std::vector<std::uint32_t>* out)
{
const size_t indexSize = format == D3DFMT_INDEX32 ? 4 : 2;
if ((size_t(start) + count) * indexSize > size)
return false;
out->resize(count);
for (UINT i = 0; i < count; ++i)
{
const uint8_t* p = bytes + (size_t(start) + i) * indexSize;
std::uint32_t index = indexSize == 4 ? (p[0] | (p[1] << 8) | (p[2] << 16) | (std::uint32_t(p[3]) << 24)) : (p[0] | (p[1] << 8));
(*out)[i] = index + base;
}
return true;
}
static void transform_point(const float v[4], const float m[16], float out[4])
{
for (int c = 0; c < 4; ++c)
out[c] = v[0] * m[c] + v[1] * m[4 + c] + v[2] * m[8 + c] + v[3] * m[12 + c];
}
static void multiply(const float a[16], const float b[16], float out[16])
{
for (int r = 0; r < 4; ++r)
transform_point(&a[r * 4], b, &out[r * 4]);
}
// The 40250 shadow pass renders a flat grey silhouette into an R5G6B5 target.
// Rasterizing that small target here keeps the D3D render-target sequence and lets
// the normal memory-texture path upload the result for terrain/object projection.
void rasterize_shadow(const Render3DDraw& draw)
{
auto* surface = static_cast<CpuSurface*>(m_renderTarget);
if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5 || draw.lines)
return;
const int width = int(surface->desc.Width), height = int(surface->desc.Height);
if (width <= 0 || height <= 0 || draw.indices.size() < 3)
return;
auto& bytes = surface->parent->levels[0].bytes;
if (bytes.size() < size_t(width) * height * 2)
return;
if (m_offscreenDepth.size() != size_t(width) * height)
m_offscreenDepth.assign(size_t(width) * height, 1.0f);
float worldView[16], mvp[16];
multiply(draw.world, draw.view, worldView);
multiply(worldView, draw.proj, mvp);
struct Point { float x, y, z; bool valid; };
std::vector<Point> points(draw.positions.size() / 3);
for (size_t i = 0; i < points.size(); ++i)
{
float p[4] = { draw.positions[i * 3], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f };
if (!draw.bone_matrices.empty() && draw.bone_weights.size() >= (i + 1) * 4 &&
draw.bone_indices.size() >= (i + 1) * 4)
{
float skinned[4] = {};
float total = 0.0f;
for (int b = 0; b < 4; ++b)
{
const float weight = draw.bone_weights[i * 4 + b];
const size_t bone = draw.bone_indices[i * 4 + b];
if (weight <= 0.0f || (bone + 1) * 16 > draw.bone_matrices.size()) continue;
float transformed[4];
transform_point(p, draw.bone_matrices.data() + bone * 16, transformed);
for (int c = 0; c < 4; ++c) skinned[c] += transformed[c] * weight;
total += weight;
}
if (total > 0.0f) { p[0] = skinned[0]; p[1] = skinned[1]; p[2] = skinned[2]; }
}
float clip[4];
transform_point(p, mvp, clip);
const bool valid = clip[3] > 0.00001f;
const float invW = valid ? 1.0f / clip[3] : 0.0f;
points[i] = { draw.viewport[0] + (clip[0] * invW * 0.5f + 0.5f) * draw.viewport[2],
draw.viewport[1] + (0.5f - clip[1] * invW * 0.5f) * draw.viewport[3],
clip[2] * invW, valid };
}
const std::uint32_t color = m_renderStates[D3DRS_TEXTUREFACTOR];
const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 |
(((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3));
auto edge = [](const Point& a, const Point& b, float x, float y) {
return (x - a.x) * (b.y - a.y) - (y - a.y) * (b.x - a.x);
};
for (size_t i = 0; i + 2 < draw.indices.size(); i += 3)
{
if (draw.indices[i] >= points.size() || draw.indices[i + 1] >= points.size() ||
draw.indices[i + 2] >= points.size()) continue;
const Point& a = points[draw.indices[i]];
const Point& b = points[draw.indices[i + 1]];
const Point& c = points[draw.indices[i + 2]];
if (!a.valid || !b.valid || !c.valid) continue;
const float area = edge(a, b, c.x, c.y);
if (std::fabs(area) < 0.00001f) continue;
const int x0 = std::max(0, int(std::floor(std::min({a.x, b.x, c.x}))));
const int y0 = std::max(0, int(std::floor(std::min({a.y, b.y, c.y}))));
const int x1 = std::min(width - 1, int(std::ceil(std::max({a.x, b.x, c.x}))));
const int y1 = std::min(height - 1, int(std::ceil(std::max({a.y, b.y, c.y}))));
// D3D top-left fill rule: a pixel centre exactly on an edge is covered only when that edge
// is a left edge (weight grows to the right) or a top edge (horizontal, weight grows downward).
auto top_left = [area](const Point& p, const Point& q) {
const float dx = (q.y - p.y) / area, dy = (p.x - q.x) / area;
return dx > 0.0f || (dx == 0.0f && dy > 0.0f);
};
const bool tlA = top_left(b, c), tlB = top_left(c, a), tlC = top_left(a, b);
auto covered = [](float w, bool tl) { return w > 0.0f || (w == 0.0f && tl); };
for (int y = y0; y <= y1; ++y)
for (int x = x0; x <= x1; ++x)
{
const float px = float(x) + 0.5f, py = float(y) + 0.5f;
const float wa = edge(b, c, px, py) / area;
const float wb = edge(c, a, px, py) / area;
const float wc = edge(a, b, px, py) / area;
if (!covered(wa, tlA) || !covered(wb, tlB) || !covered(wc, tlC)) continue;
const float z = wa * a.z + wb * b.z + wc * c.z;
const size_t pixel = size_t(y) * width + x;
if (z < 0.0f || z > 1.0f || z > m_offscreenDepth[pixel]) continue;
m_offscreenDepth[pixel] = z;
std::memcpy(bytes.data() + pixel * 2, &rgb565, 2);
}
}
}
// Each two-triangle quad of a UI-space draw becomes an image command: its screen corners through
// world * view * projection, its stage-0 texture (or D3DTA_TFACTOR colour when stage 0 selects it)
// and, when stage 1 generates coordinates from the camera-space position through D3DTS_TEXTURE1
// (the minimap's circular filter), the mask's coordinates at the corners.
void record_ui_quads(D3DPRIMITIVETYPE type, const uint8_t* vertices, size_t vertexBytes, UINT stride,
const VertexLayout& layout, const std::vector<std::uint32_t>& indices)
{
if (layout.uv0 < 0)
return;
unsigned width = 0, height = 0;
UIRenderGetSize(&width, &height);
float worldView[16], worldViewProj[16];
multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView);
multiply(worldView, m_transforms[D3DTS_PROJECTION], worldViewProj);
const DWORD* stage0 = m_stageStates[0];
const DWORD* stage1 = m_stageStates[1];
const bool factorColor = stage0[D3DTSS_COLOROP] == D3DTOP_SELECTARG1 && stage0[D3DTSS_COLORARG1] == D3DTA_TFACTOR;
const std::uint32_t argb = factorColor ? (0xFF000000u | (m_renderStates[D3DRS_TEXTUREFACTOR] & 0x00FFFFFFu)) : 0xFFFFFFFFu;
const bool masked = m_textures[1] && (stage1[D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == D3DTSS_TCI_CAMERASPACEPOSITION &&
stage1[D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2 && stage1[D3DTSS_COLOROP] == D3DTOP_MODULATE &&
stage1[D3DTSS_ALPHAOP] == D3DTOP_SELECTARG1 && stage1[D3DTSS_ALPHAARG1] == D3DTA_TEXTURE;
const std::string texture = factorColor ? std::string() : UIRenderTextureNameFromHandle(m_textures[0]);
const std::string mask = masked ? UIRenderTextureNameFromHandle(m_textures[1]) : std::string();
std::vector<std::vector<std::uint32_t>> quads;
if (type == D3DPT_TRIANGLELIST)
for (size_t i = 0; i + 6 <= indices.size(); i += 6)
quads.push_back({ indices.begin() + i, indices.begin() + i + 6 });
else if (type == D3DPT_TRIANGLESTRIP && indices.size() == 4)
quads.push_back(indices);
for (const auto& quad : quads)
{
std::vector<std::uint32_t> corners;
for (std::uint32_t index : quad)
if (std::find(corners.begin(), corners.end(), index) == corners.end())
corners.push_back(index);
if (corners.size() != 4)
continue;
float sx[4], sy[4], u[4], v[4], mu[4] = {}, mv[4] = {};
bool inside = true;
for (int i = 0; i < 4; ++i)
{
const size_t offset = size_t(corners[i]) * stride;
if (offset + stride > vertexBytes) { inside = false; break; }
const uint8_t* vertex = vertices + offset;
float position[4] = { 0, 0, 0, 1 }, clip[4], camera[4], coord[4];
std::memcpy(position, vertex, 12);
std::memcpy(&u[i], vertex + layout.uv0, 4);
std::memcpy(&v[i], vertex + layout.uv0 + 4, 4);
transform_point(position, worldViewProj, clip);
if (clip[3] == 0.0f) { inside = false; break; }
sx[i] = (clip[0] / clip[3] * 0.5f + 0.5f) * float(width);
sy[i] = (0.5f - clip[1] / clip[3] * 0.5f) * float(height);
if (masked)
{
transform_point(position, worldView, camera);
camera[3] = 1.0f;
transform_point(camera, m_transforms[D3DTS_TEXTURE1], coord);
mu[i] = coord[0];
mv[i] = coord[1];
}
}
if (!inside)
continue;
UIRenderCommand command{UIRenderCommand::Image, 0, 0, 0, 0, argb};
command.su = *std::min_element(u, u + 4); command.eu = *std::max_element(u, u + 4);
command.sv = *std::min_element(v, v + 4); command.ev = *std::max_element(v, v + 4);
// Corners in the command's TL, TR, BL, BR order, by texture coordinate.
const float cu[4] = { command.su, command.eu, command.su, command.eu };
const float cv[4] = { command.sv, command.sv, command.ev, command.ev };
for (int c = 0; c < 4; ++c)
{
int best = 0;
float bestDistance = INFINITY;
for (int i = 0; i < 4; ++i)
{
const float distance = std::fabs(u[i] - cu[c]) + std::fabs(v[i] - cv[c]);
if (distance < bestDistance) { bestDistance = distance; best = i; }
}
command.qx[c] = sx[best]; command.qy[c] = sy[best];
command.mu[c] = mu[best]; command.mv[c] = mv[best];
}
command.x1 = *std::min_element(command.qx, command.qx + 4);
command.y1 = *std::min_element(command.qy, command.qy + 4);
command.x2 = *std::max_element(command.qx, command.qx + 4);
command.y2 = *std::max_element(command.qy, command.qy + 4);
command.text = texture;
command.quad = true;
command.mask = mask;
UIRenderAdd(std::move(command));
}
}
void record(D3DPRIMITIVETYPE type, UINT primitives, const uint8_t* vertices, size_t vertexBytes, UINT stride,
const std::vector<std::uint32_t>& indices, const MtCpuVertexBuffer* source_vertex = nullptr,
const MtCpuIndexBuffer* source_index = nullptr, UINT source_first_index = 0)
{
if (type == D3DPT_POINTLIST || indices.empty() || !vertices)
return;
VertexLayout layout = fvf_layout(m_fvf);
if (!stride)
stride = layout.stride;
if (!stride || stride < (layout.rhw ? 16u : 12u))
return;
// Elements past the stream's stride live in another stream (CreatePTStreamVertexShader).
if (layout.normal + 12 > int(stride)) layout.normal = -1;
if (layout.diffuse + 4 > int(stride)) layout.diffuse = -1;
if (layout.uv0 + 8 > int(stride)) layout.uv0 = -1;
if (layout.uv1 + 8 > int(stride)) layout.uv1 = -1;
// An orthographic projection over untransformed vertices is a UI-space draw (CPythonMiniMap's
// terrain tiles under CPythonGraphic::SetOrtho2D): it joins the UI command stream in order.
if (m_renderTarget == m_backBuffer && !layout.rhw && m_transforms[D3DTS_PROJECTION][11] == 0.0f)
{
record_ui_quads(type, vertices, vertexBytes, stride, layout, indices);
return;
}
Render3DDraw draw;
std::memcpy(draw.world, m_transforms[256], sizeof(draw.world)); // D3DTS_WORLD
std::memcpy(draw.view, m_transforms[D3DTS_VIEW], sizeof(draw.view));
std::memcpy(draw.proj, m_transforms[D3DTS_PROJECTION], sizeof(draw.proj));
draw.texture0 = UIRenderTextureNameFromHandle(m_textures[0]);
draw.texture1 = UIRenderTextureNameFromHandle(m_textures[1]);
draw.viewport[0] = float(m_viewport.X);
draw.viewport[1] = float(m_viewport.Y);
draw.viewport[2] = float(m_viewport.Width);
draw.viewport[3] = float(m_viewport.Height);
draw.viewport_z[0] = m_viewport.MinZ;
draw.viewport_z[1] = m_viewport.MaxZ;
draw.pretransformed = layout.rhw;
draw.lines = type == D3DPT_LINELIST || type == D3DPT_LINESTRIP;
// Rebase: copy only the vertices the indices reference.
std::uint32_t lo = indices[0], hi = indices[0];
for (std::uint32_t index : indices)
{
lo = std::min(lo, index);
hi = std::max(hi, index);
}
if ((size_t(hi) + 1) * stride > vertexBytes)
return;
const size_t count = size_t(hi - lo) + 1;
GpuSkinSubrangeView skin_view;
const bool gpu_skinned = source_vertex != nullptr &&
LookupGpuSkinSubrange(vertices, stride, lo, hi, &skin_view);
if (source_vertex)
{
// The original buffers outlive individual draw calls. Keep their identity separate from
// Unlock revisions so a static surface can keep its Godot mesh across frames.
auto mix = [](std::uint64_t hash, std::uint64_t value) {
return (hash ^ value) * 1099511628211ull;
};
std::uint64_t key = 14695981039346656037ull;
if (gpu_skinned)
{
key = mix(key, skin_view.source_mesh_key);
key = mix(key, source_index ? source_index->id : 0);
key = mix(key, source_first_index);
key = mix(key, lo - skin_view.mesh_base_vertex);
key = mix(key, hi - skin_view.mesh_base_vertex);
key = mix(key, primitives);
key = mix(key, type);
key = mix(key, stride);
key = mix(key, m_fvf);
draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull;
draw.geometry_revision = 1ull;
}
else
{
key = mix(key, source_vertex->id);
key = mix(key, source_index ? source_index->id : 0);
key = mix(key, source_first_index);
key = mix(key, lo);
key = mix(key, hi);
key = mix(key, primitives);
key = mix(key, type);
key = mix(key, stride);
key = mix(key, m_fvf);
draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull;
std::uint64_t revision = mix(14695981039346656037ull, source_vertex->revision);
revision = mix(revision, source_index ? source_index->revision : 0);
draw.geometry_revision = revision & 0x7FFFFFFFFFFFFFFFull;
}
}
const std::uint64_t frame_id = UIRenderFrameId();
auto cached = draw.geometry_key ? m_geometry_cache.find(draw.geometry_key) : m_geometry_cache.end();
const bool reuse = cached != m_geometry_cache.end() && !cached->second.changing &&
cached->second.revision == draw.geometry_revision;
if (reuse)
{
cached->second.last_used = frame_id;
const Render3DDraw& prior = cached->second.geometry;
draw.positions = prior.positions;
draw.rhw = prior.rhw;
draw.normals = prior.normals;
draw.uv0 = prior.uv0;
draw.uv1 = prior.uv1;
draw.diffuse = prior.diffuse;
draw.indices = prior.indices;
draw.bone_indices = prior.bone_indices;
draw.bone_weights = prior.bone_weights;
}
else
{
draw.positions.resize(count * 3);
if (layout.rhw) draw.rhw.resize(count);
if (layout.normal >= 0) draw.normals.resize(count * 3);
if (layout.uv0 >= 0) draw.uv0.resize(count * 2);
if (layout.uv1 >= 0) draw.uv1.resize(count * 2);
if (layout.diffuse >= 0) draw.diffuse.resize(count);
for (size_t i = 0; i < count; ++i)
{
const uint8_t* v = vertices + (lo + i) * stride;
std::memcpy(&draw.positions[i * 3], v, 12);
if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4);
if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12);
if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8);
if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8);
if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4);
}
if (gpu_skinned)
{
const std::size_t rel_lo = std::size_t(lo - skin_view.mesh_base_vertex);
draw.bone_indices.resize(count * 4);
draw.bone_weights.resize(count * 4);
std::memcpy(draw.bone_indices.data(), skin_view.bone_indices + rel_lo * 4, count * 4 * sizeof(std::uint8_t));
std::memcpy(draw.bone_weights.data(), skin_view.bone_weights + rel_lo * 4, count * 4 * sizeof(float));
}
// Strips and fans become lists; D3D flips the winding of every odd strip triangle.
switch (type)
{
case D3DPT_TRIANGLESTRIP:
for (UINT i = 0; i < primitives; ++i)
{
const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo;
if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c });
else draw.indices.insert(draw.indices.end(), { a, b, c });
}
break;
case D3DPT_TRIANGLEFAN:
for (UINT i = 0; i < primitives; ++i)
draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo });
break;
case D3DPT_LINESTRIP:
for (UINT i = 0; i < primitives; ++i)
draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo });
break;
default:
for (std::uint32_t index : indices)
draw.indices.push_back(index - lo);
break;
}
if (cached != m_geometry_cache.end())
{
cached->second.last_used = frame_id;
if (!cached->second.changing && cached->second.revision != draw.geometry_revision)
{
cached->second.changing = true;
cached->second.geometry = Render3DDraw();
}
}
else if (draw.geometry_key && cached == m_geometry_cache.end())
m_geometry_cache.emplace(draw.geometry_key, GeometryCacheEntry{draw.geometry_revision, frame_id, false, draw});
}
if (gpu_skinned && skin_view.bone_matrices && skin_view.bone_count > 0)
draw.bone_matrices.assign(skin_view.bone_matrices, skin_view.bone_matrices + std::size_t(skin_view.bone_count) * 16);
if (frame_id >= m_cache_pruned_frame + 120)
{
m_cache_pruned_frame = frame_id;
for (auto it = m_geometry_cache.begin(); it != m_geometry_cache.end();)
it = frame_id - it->second.last_used > 120 ? m_geometry_cache.erase(it) : std::next(it);
}
draw.alpha_blend = m_renderStates[D3DRS_ALPHABLENDENABLE];
draw.src_blend = m_renderStates[D3DRS_SRCBLEND];
draw.dest_blend = m_renderStates[D3DRS_DESTBLEND];
draw.alpha_test = m_renderStates[D3DRS_ALPHATESTENABLE];
draw.alpha_ref = m_renderStates[D3DRS_ALPHAREF];
draw.alpha_func = m_renderStates[D3DRS_ALPHAFUNC];
draw.cull_mode = m_renderStates[D3DRS_CULLMODE];
draw.z_enable = m_renderStates[D3DRS_ZENABLE];
draw.z_write = m_renderStates[D3DRS_ZWRITEENABLE];
draw.z_func = m_renderStates[D3DRS_ZFUNC];
draw.lighting = m_renderStates[D3DRS_LIGHTING];
draw.texture_factor = m_renderStates[D3DRS_TEXTUREFACTOR];
draw.fog_enable = m_renderStates[D3DRS_FOGENABLE];
draw.fog_color = m_renderStates[D3DRS_FOGCOLOR];
draw.fog_vertex_mode = m_renderStates[D3DRS_FOGVERTEXMODE];
draw.fog_table_mode = m_renderStates[D3DRS_FOGTABLEMODE];
draw.fog_range_enable = m_renderStates[D3DRS_RANGEFOGENABLE];
std::memcpy(&draw.fog_start, &m_renderStates[D3DRS_FOGSTART], sizeof(float));
std::memcpy(&draw.fog_end, &m_renderStates[D3DRS_FOGEND], sizeof(float));
std::memcpy(&draw.fog_density, &m_renderStates[D3DRS_FOGDENSITY], sizeof(float));
draw.ambient = m_renderStates[D3DRS_AMBIENT];
draw.diffuse_material_source = m_renderStates[D3DRS_DIFFUSEMATERIALSOURCE];
draw.ambient_material_source = m_renderStates[D3DRS_AMBIENTMATERIALSOURCE];
draw.emissive_material_source = m_renderStates[D3DRS_EMISSIVEMATERIALSOURCE];
draw.color_vertex = m_renderStates[D3DRS_COLORVERTEX];
draw.normalize_normals = m_renderStates[D3DRS_NORMALIZENORMALS];
draw.local_viewer = m_renderStates[D3DRS_LOCALVIEWER];
for (int stage = 0; stage < 2; ++stage)
{
draw.color_op[stage] = m_stageStates[stage][D3DTSS_COLOROP];
draw.color_arg1[stage] = m_stageStates[stage][D3DTSS_COLORARG1];
draw.color_arg2[stage] = m_stageStates[stage][D3DTSS_COLORARG2];
draw.alpha_op[stage] = m_stageStates[stage][D3DTSS_ALPHAOP];
draw.alpha_arg1[stage] = m_stageStates[stage][D3DTSS_ALPHAARG1];
draw.alpha_arg2[stage] = m_stageStates[stage][D3DTSS_ALPHAARG2];
draw.address_u[stage] = m_stageStates[stage][D3DTSS_ADDRESSU];
draw.address_v[stage] = m_stageStates[stage][D3DTSS_ADDRESSV];
draw.min_filter[stage] = m_stageStates[stage][D3DTSS_MINFILTER];
draw.mag_filter[stage] = m_stageStates[stage][D3DTSS_MAGFILTER];
draw.mip_filter[stage] = m_stageStates[stage][D3DTSS_MIPFILTER];
draw.border_color[stage] = m_stageStates[stage][D3DTSS_BORDERCOLOR];
draw.texcoord_index[stage] = m_stageStates[stage][D3DTSS_TEXCOORDINDEX];
draw.texture_transform_flags[stage] = m_stageStates[stage][D3DTSS_TEXTURETRANSFORMFLAGS];
std::memcpy(draw.texture_matrix[stage], m_transforms[D3DTS_TEXTURE0 + stage], sizeof(draw.texture_matrix[stage]));
}
copy_color(draw.material_diffuse, m_material.Diffuse);
copy_color(draw.material_ambient, m_material.Ambient);
copy_color(draw.material_emissive, m_material.Emissive);
if (m_lightEnabled[0] && (m_lights[0].Type == D3DLIGHT_DIRECTIONAL || m_lights[0].Type == D3DLIGHT_SPOT))
{
draw.light0 = true;
draw.light0_direction[0] = m_lights[0].Direction.x;
draw.light0_direction[1] = m_lights[0].Direction.y;
draw.light0_direction[2] = m_lights[0].Direction.z;
copy_color(draw.light0_diffuse, m_lights[0].Diffuse);
copy_color(draw.light0_ambient, m_lights[0].Ambient);
}
for (int i = 0; i < 8; ++i)
{
if (!m_lightEnabled[i]) continue;
const D3DLIGHT8& source = m_lights[i];
auto& light = draw.lights[i];
light.type = source.Type;
light.position[0] = source.Position.x;
light.position[1] = source.Position.y;
light.position[2] = source.Position.z;
light.direction[0] = source.Direction.x;
light.direction[1] = source.Direction.y;
light.direction[2] = source.Direction.z;
copy_color(light.diffuse, source.Diffuse);
copy_color(light.ambient, source.Ambient);
light.attenuation[0] = source.Attenuation0;
light.attenuation[1] = source.Attenuation1;
light.attenuation[2] = source.Attenuation2;
light.range = source.Range;
light.theta = source.Theta;
light.phi = source.Phi;
light.falloff = source.Falloff;
}
if (m_renderTarget == m_backBuffer)
Render3DAdd(std::move(draw));
else
rasterize_shadow(draw);
}
ULONG m_refs = 1;
int m_width, m_height;
D3DVIEWPORT8 m_viewport = {};
float m_transforms[kTransforms][16];
DWORD m_renderStates[kRenderStates] = {};
DWORD m_stageStates[kStages][kStageStates] = {};
IDirect3DBaseTexture8* m_textures[kStages] = {};
D3DMATERIAL8 m_material;
D3DLIGHT8 m_lights[8];
bool m_lightEnabled[8] = {};
IDirect3DSurface8* m_backBuffer = nullptr;
IDirect3DSurface8* m_depthBuffer = nullptr;
IDirect3DSurface8* m_renderTarget = nullptr;
IDirect3DSurface8* m_depthStencil = nullptr;
DWORD m_fvf = 0;
MtCpuVertexBuffer* m_stream = nullptr;
UINT m_streamStride = 0;
MtCpuIndexBuffer* m_indices = nullptr;
UINT m_baseVertex = 0;
struct GeometryCacheEntry {
std::uint64_t revision;
std::uint64_t last_used;
bool changing;
Render3DDraw geometry;
};
std::unordered_map<std::uint64_t, GeometryCacheEntry> m_geometry_cache;
std::uint64_t m_cache_pruned_frame = 0;
std::vector<float> m_offscreenDepth;
};
}
std::string MtCpuTextureNameFromHandle(const IDirect3DBaseTexture8* handle)
{
if (!handle) return {};
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
for (const auto& entry : g_cpu_textures)
{
if (entry.second == handle && !entry.second->levels.empty())
return "mem:cpu_" + std::to_string(entry.first) + "@" + std::to_string(entry.second->revision);
}
return {};
}
bool MtCpuMemoryTexture(const std::string& name, UIMemoryTexture* out)
{
if (name.compare(0, 8, "mem:cpu_") != 0 || !out) return false;
const unsigned id = unsigned(std::strtoul(name.c_str() + 8, nullptr, 10));
std::lock_guard<std::mutex> lock(g_cpu_tex_mutex);
auto it = g_cpu_textures.find(id);
if (it == g_cpu_textures.end() || it->second->levels.empty()) return false;
const auto& lv0 = it->second->levels[0];
const UINT w = lv0.desc.Width;
const UINT h = lv0.desc.Height;
if (!w || !h) return false;
out->width = static_cast<int>(w);
out->height = static_cast<int>(h);
out->revision = it->second->revision;
out->argb.resize(std::size_t(w) * std::size_t(h));
const uint8_t* bytes = lv0.bytes.data();
const int bpp = format_bytes(lv0.desc.Format);
for (std::size_t i = 0; i < out->argb.size(); ++i)
{
if (bpp == 4)
{
std::uint32_t px = 0;
std::memcpy(&px, bytes + i * 4, 4);
// CTerrain::PutImage32 packs alpha into bits 31..24 with RGB == 0.
// Set RGB to white so stage-1 modulation preserves the base texture's RGB.
if ((px & 0x00FFFFFFu) == 0)
px |= 0x00FFFFFFu;
out->argb[i] = px;
}
else if (lv0.desc.Format == D3DFMT_R5G6B5)
{
std::uint16_t word = 0;
std::memcpy(&word, bytes + i * 2, 2);
const std::uint32_t r = ((word >> 11) & 31u) * 255u / 31u;
const std::uint32_t g = ((word >> 5) & 63u) * 255u / 63u;
const std::uint32_t b = (word & 31u) * 255u / 31u;
out->argb[i] = 0xFF000000u | (r << 16) | (g << 8) | b;
}
else if (bpp == 2)
{
std::uint16_t word = 0;
std::memcpy(&word, bytes + i * 2, 2);
// CTerrain::PutImage16 writes `src[x] << 8`, storing the full 8-bit alpha in the high byte.
const std::uint32_t a = (word >> 8) & 0xFFu;
out->argb[i] = (a << 24) | 0x00FFFFFFu;
}
else
{
const std::uint32_t a = bytes[i];
out->argb[i] = (a << 24) | 0x00FFFFFFu;
}
}
return true;
}
IDirect3DDevice8* MtCreateRecordingDevice(int width, int height) { return new RecordingDevice(width, height); }
void SetNativeTerrainRenderEnabled(bool enabled)
{
g_native_terrain_override = enabled ? 1 : 0;
}
bool IsNativeTerrainRenderEnabled()
{
if (g_native_terrain_override >= 0)
return g_native_terrain_override != 0;
static const bool env_on = [] {
const char* v = std::getenv("MT_NATIVE_TERRAIN");
return v && (*v == '1' || *v == 't' || *v == 'T' || *v == 'y' || *v == 'Y');
}();
return env_on;
}
void Render3DBeginFrame()
{
std::lock_guard<std::mutex> lock(g_draws_mutex);
g_draws.clear();
}
void Render3DAdd(Render3DDraw draw)
{
std::lock_guard<std::mutex> lock(g_draws_mutex);
g_draws.push_back(std::move(draw));
}
const std::vector<Render3DDraw>& Render3DDraws() { return g_draws; }
// D3D8 fixed-function texture coordinate processing for one stage (the native renderer's
// native.vert applies the same rules on the GPU):
// - D3DTSS_TEXCOORDINDEX low word picks the vertex set, a 2D set enters the transform as (u, v, 1, 0)
// (so a D3DTS_TEXTUREn translation lives in _31/_32, as CSkyBox::RenderCloud writes it);
// - D3DTSS_TCI_CAMERASPACENORMAL/POSITION/REFLECTIONVECTOR generate (x, y, z, 1) in camera space;
// - D3DTSS_TEXTURETRANSFORMFLAGS != DISABLE multiplies by D3DTS_TEXTUREn (row vector), and
// D3DTTFF_PROJECTED divides by the last counted element.
void Render3DStageTexcoords(const Render3DDraw& draw, int stage, std::vector<float>& out)
{
const size_t count = draw.positions.size() / 3;
out.assign(count * 2, 0.0f);
if (stage < 0 || stage > 1)
return;
const std::uint32_t tci = draw.texcoord_index[stage];
const std::uint32_t gen = tci & 0xFFFF0000u;
const std::uint32_t set = tci & 0xFFFFu;
const std::vector<float>* uv = set == 0 ? &draw.uv0 : (set == 1 ? &draw.uv1 : nullptr);
const std::uint32_t flags = draw.texture_transform_flags[stage];
const std::uint32_t elements = flags & 0xFFu;
const bool projected = (flags & D3DTTFF_PROJECTED) != 0;
float worldView[16] = {};
for (int row = 0; row < 4; ++row)
for (int col = 0; col < 4; ++col)
for (int k = 0; k < 4; ++k)
worldView[row * 4 + col] += draw.world[row * 4 + k] * draw.view[k * 4 + col];
// Normals go through the inverse transpose of world * view (cofactors / determinant).
const float* w = worldView;
float normalMatrix[9] = {
w[5] * w[10] - w[6] * w[9], -(w[4] * w[10] - w[6] * w[8]), w[4] * w[9] - w[5] * w[8],
-(w[1] * w[10] - w[2] * w[9]), w[0] * w[10] - w[2] * w[8], -(w[0] * w[9] - w[1] * w[8]),
w[1] * w[6] - w[2] * w[5], -(w[0] * w[6] - w[2] * w[4]), w[0] * w[5] - w[1] * w[4] };
const float det = w[0] * normalMatrix[0] + w[1] * normalMatrix[1] + w[2] * normalMatrix[2];
if (det != 0.0f)
for (float& c : normalMatrix) c /= det;
const float* m = draw.texture_matrix[stage];
for (size_t i = 0; i < count; ++i)
{
float in[4] = { 0, 0, 1, 0 };
if (gen == 0 || draw.pretransformed)
{
if (uv && uv->size() >= (i + 1) * 2)
{
in[0] = (*uv)[i * 2 + 0];
in[1] = (*uv)[i * 2 + 1];
}
}
else
{
const float* p = &draw.positions[i * 3];
float eye[3], n[3] = { 0, 0, 0 };
for (int c = 0; c < 3; ++c)
eye[c] = p[0] * worldView[c] + p[1] * worldView[4 + c] + p[2] * worldView[8 + c] + worldView[12 + c];
if (draw.normals.size() >= (i + 1) * 3)
{
const float* src = &draw.normals[i * 3];
for (int c = 0; c < 3; ++c)
n[c] = src[0] * normalMatrix[c] + src[1] * normalMatrix[3 + c] + src[2] * normalMatrix[6 + c];
if (draw.normalize_normals)
{
const float len = std::sqrt(n[0] * n[0] + n[1] * n[1] + n[2] * n[2]);
if (len > 0.0f)
for (float& c : n) c /= len;
}
}
float v[3] = { 0, 0, 0 };
if (gen == D3DTSS_TCI_CAMERASPACENORMAL)
std::memcpy(v, n, sizeof(v));
else if (gen == D3DTSS_TCI_CAMERASPACEPOSITION)
std::memcpy(v, eye, sizeof(v));
else if (gen == D3DTSS_TCI_CAMERASPACEREFLECTIONVECTOR)
{
float e[3] = { 0, 0, -1 };
if (draw.local_viewer)
{
const float len = std::sqrt(eye[0] * eye[0] + eye[1] * eye[1] + eye[2] * eye[2]);
for (int c = 0; c < 3; ++c) e[c] = len > 0.0f ? -eye[c] / len : 0.0f;
}
const float en = e[0] * n[0] + e[1] * n[1] + e[2] * n[2];
for (int c = 0; c < 3; ++c) v[c] = 2.0f * en * n[c] - e[c];
}
in[0] = v[0]; in[1] = v[1]; in[2] = v[2]; in[3] = 1.0f;
}
float r[4] = { in[0], in[1], in[2], in[3] };
// Pre-transformed (XYZRHW) vertices pass their coordinates through untransformed.
if (elements != D3DTTFF_DISABLE && !draw.pretransformed)
for (int c = 0; c < 4; ++c)
r[c] = in[0] * m[c] + in[1] * m[4 + c] + in[2] * m[8 + c] + in[3] * m[12 + c];
if (projected && !draw.pretransformed && elements >= 2 && elements <= 4 && r[elements - 1] != 0.0f)
{
r[0] /= r[elements - 1];
r[1] /= r[elements - 1];
}
out[i * 2 + 0] = r[0];
out[i * 2 + 1] = r[1];
}
}