#include "RecordingDevice.h" #include "RenderCommands3D.h" #include "UIRenderCommands.h" #include "CpuBuffer.h" #include #include #include #include #include #include #include namespace { std::mutex g_draws_mutex; std::vector g_draws; int g_native_terrain_override = -1; struct VertexLayout { unsigned stride = 0; bool rhw = false; int normal = -1, diffuse = -1, uv0 = -1, uv1 = -1; }; // The element offsets of a fixed-function FVF (D3DXGetFVFVertexSize's order). VertexLayout fvf_layout(DWORD fvf) { VertexLayout layout; unsigned offset = 0; switch (fvf & D3DFVF_POSITION_MASK) { case D3DFVF_XYZ: offset = 12; break; case D3DFVF_XYZRHW: offset = 16; layout.rhw = true; break; case D3DFVF_XYZB1: offset = 16; break; case D3DFVF_XYZB2: offset = 20; break; case D3DFVF_XYZB3: offset = 24; break; case D3DFVF_XYZB4: offset = 28; break; case D3DFVF_XYZB5: offset = 32; break; } if (fvf & D3DFVF_NORMAL) { layout.normal = offset; offset += 12; } if (fvf & D3DFVF_PSIZE) offset += 4; if (fvf & D3DFVF_DIFFUSE) { layout.diffuse = offset; offset += 4; } if (fvf & D3DFVF_SPECULAR) offset += 4; const unsigned texCount = (fvf & D3DFVF_TEXCOUNT_MASK) >> D3DFVF_TEXCOUNT_SHIFT; for (unsigned i = 0; i < texCount; ++i) { static const unsigned coordSize[4] = { 8, 12, 16, 4 }; if (i == 0) layout.uv0 = offset; if (i == 1) layout.uv1 = offset; offset += coordSize[(fvf >> (16 + i * 2)) & 3]; } layout.stride = offset; return layout; } void copy_color(float out[4], const D3DCOLORVALUE& c) { out[0] = c.r; out[1] = c.g; out[2] = c.b; out[3] = c.a; } int format_bytes(D3DFORMAT format) { switch (format) { case D3DFMT_A4R4G4B4: case D3DFMT_R5G6B5: case D3DFMT_A1R5G5B5: case D3DFMT_X1R5G5B5: case D3DFMT_X4R4G4B4: case D3DFMT_D16: case D3DFMT_D16_LOCKABLE: return 2; case D3DFMT_A8: case D3DFMT_L8: return 1; default: return 4; } } // PORT: device-created resources live in CPU memory with D3D reference counting. Textures made by // CreateTexture resources keep their mip levels as bytes. Terrain alpha maps and the character // shadow render target are sampled by the native renderer through MtCpuMemoryTexture. struct CpuTexture; struct CpuSurface final : IDirect3DSurface8 { ULONG refs = 1; CpuTexture* parent = nullptr; // a texture level, or null for a standalone surface UINT level = 0; D3DSURFACE_DESC desc = {}; std::vector bytes; // standalone surfaces only ULONG AddRef() override { return ++refs; } ULONG Release() override; HRESULT GetDesc(D3DSURFACE_DESC* out) override { *out = desc; return S_OK; } HRESULT LockRect(D3DLOCKED_RECT* locked, const RECT*, DWORD) override; HRESULT UnlockRect() override { return S_OK; } }; struct CpuTexture; std::mutex g_cpu_tex_mutex; std::unordered_map g_cpu_textures; unsigned g_next_cpu_tex_id = 1; struct CpuTexture final : IDirect3DTexture8 { struct Level { D3DSURFACE_DESC desc; std::vector bytes; }; unsigned id = 0; std::uint32_t revision = 1; ULONG refs = 1; std::vector levels; CpuTexture() { std::lock_guard lock(g_cpu_tex_mutex); id = g_next_cpu_tex_id++; g_cpu_textures[id] = this; } ~CpuTexture() { std::lock_guard lock(g_cpu_tex_mutex); g_cpu_textures.erase(id); } ULONG AddRef() override { return ++refs; } ULONG Release() override { const ULONG left = --refs; if (!left) delete this; return left; } DWORD GetLevelCount() override { return DWORD(levels.size()); } HRESULT GetLevelDesc(UINT level, D3DSURFACE_DESC* out) override { if (level >= levels.size()) return E_FAIL; *out = levels[level].desc; return S_OK; } HRESULT GetSurfaceLevel(UINT level, IDirect3DSurface8** out) override { if (level >= levels.size()) return E_FAIL; auto* surface = new CpuSurface(); surface->parent = this; surface->level = level; surface->desc = levels[level].desc; AddRef(); *out = surface; return S_OK; } HRESULT LockRect(UINT level, D3DLOCKED_RECT* locked, const RECT*, DWORD) override { if (level >= levels.size()) return E_FAIL; locked->Pitch = INT(levels[level].desc.Width * format_bytes(levels[level].desc.Format)); locked->pBits = levels[level].bytes.data(); return S_OK; } HRESULT UnlockRect(UINT level) override { if (level >= levels.size()) return E_FAIL; if (level == 0) ++revision; return S_OK; } }; ULONG CpuSurface::Release() { const ULONG left = --refs; if (!left) { if (parent) parent->Release(); delete this; } return left; } HRESULT CpuSurface::LockRect(D3DLOCKED_RECT* locked, const RECT* rect, DWORD flags) { if (parent) return parent->LockRect(level, locked, rect, flags); locked->Pitch = INT(desc.Width * format_bytes(desc.Format)); locked->pBits = bytes.data(); return S_OK; } D3DSURFACE_DESC surface_desc(UINT width, UINT height, D3DFORMAT format, DWORD usage, D3DPOOL pool) { D3DSURFACE_DESC desc = {}; desc.Format = format; desc.Type = D3DRTYPE_SURFACE; desc.Usage = usage; desc.Pool = pool; desc.Size = width * height * format_bytes(format); desc.MultiSampleType = D3DMULTISAMPLE_NONE; desc.Width = width; desc.Height = height; return desc; } struct DeviceVertexBuffer final : MtCpuVertexBuffer { ULONG refs = 1; ULONG AddRef() override { return ++refs; } ULONG Release() override { const ULONG left = --refs; if (!left) delete this; return left; } }; struct DeviceIndexBuffer final : MtCpuIndexBuffer { ULONG refs = 1; ULONG AddRef() override { return ++refs; } ULONG Release() override { const ULONG left = --refs; if (!left) delete this; return left; } }; class RecordingDevice final : public IDirect3DDevice8 { public: RecordingDevice(int width, int height) : m_width(width), m_height(height) { static const float identity[16] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; for (auto& matrix : m_transforms) std::memcpy(matrix, identity, sizeof(identity)); std::memset(&m_material, 0, sizeof(m_material)); m_material.Diffuse = { 1, 1, 1, 1 }; std::memset(m_lights, 0, sizeof(m_lights)); auto* backBuffer = new CpuSurface(); backBuffer->desc = surface_desc(width, height, D3DFMT_X8R8G8B8, D3DUSAGE_RENDERTARGET, D3DPOOL_DEFAULT); m_backBuffer = backBuffer; auto* depthBuffer = new CpuSurface(); depthBuffer->desc = surface_desc(width, height, D3DFMT_D16, D3DUSAGE_DEPTHSTENCIL, D3DPOOL_DEFAULT); m_depthBuffer = depthBuffer; m_renderTarget = m_backBuffer; m_renderTarget->AddRef(); m_depthStencil = m_depthBuffer; m_depthStencil->AddRef(); m_viewport = { 0, 0, (DWORD)width, (DWORD)height, 0.0f, 1.0f }; m_renderStates[D3DRS_TEXTUREFACTOR] = 0xFFFFFFFF; m_renderStates[D3DRS_ZENABLE] = TRUE; m_renderStates[D3DRS_ZWRITEENABLE] = TRUE; m_renderStates[D3DRS_CULLMODE] = D3DCULL_CCW; m_renderStates[D3DRS_LIGHTING] = TRUE; m_renderStates[D3DRS_SRCBLEND] = D3DBLEND_ONE; m_renderStates[D3DRS_DESTBLEND] = D3DBLEND_ZERO; m_renderStates[D3DRS_ALPHAFUNC] = D3DCMP_ALWAYS; m_stageStates[0][D3DTSS_COLOROP] = D3DTOP_MODULATE; m_stageStates[0][D3DTSS_COLORARG1] = D3DTA_TEXTURE; m_stageStates[0][D3DTSS_COLORARG2] = D3DTA_CURRENT; m_stageStates[0][D3DTSS_ALPHAOP] = D3DTOP_SELECTARG1; m_stageStates[0][D3DTSS_ALPHAARG1] = D3DTA_TEXTURE; m_stageStates[1][D3DTSS_COLOROP] = D3DTOP_DISABLE; m_stageStates[1][D3DTSS_ALPHAOP] = D3DTOP_DISABLE; } ~RecordingDevice() override { m_renderTarget->Release(); m_depthStencil->Release(); m_backBuffer->Release(); m_depthBuffer->Release(); } ULONG AddRef() override { return ++m_refs; } ULONG Release() override { const ULONG refs = --m_refs; if (!refs) delete this; return refs; } // PORT: a D3D8 HAL on a T&L card (D3DDEVCAPS_HWTRANSFORMANDLIGHT), 8 lights, 4 texture stages. HRESULT GetDeviceCaps(D3DCAPS8* caps) override { std::memset(caps, 0, sizeof(*caps)); caps->DevCaps = D3DDEVCAPS_HWTRANSFORMANDLIGHT; caps->MaxActiveLights = 8; caps->MaxTextureBlendStages = 4; caps->MaxSimultaneousTextures = 4; caps->MaxTextureWidth = caps->MaxTextureHeight = 4096; caps->MaxAnisotropy = 16; caps->MaxPrimitiveCount = 0xFFFFF; caps->MaxVertexIndex = 0xFFFFF; caps->MaxStreams = 8; caps->MaxStreamStride = 255; caps->TextureAddressCaps = D3DPTADDRESSCAPS_BORDER | D3DPTADDRESSCAPS_CLAMP | D3DPTADDRESSCAPS_WRAP | D3DPTADDRESSCAPS_MIRROR; return S_OK; } HRESULT GetViewport(D3DVIEWPORT8* viewport) override { if (viewport) *viewport = m_viewport; return S_OK; } HRESULT SetViewport(const D3DVIEWPORT8* viewport) override { if (viewport) m_viewport = *viewport; return S_OK; } // PORT: CPU memory has no texture budget; report the 64MB of a mid-range 2004 card. UINT GetAvailableTextureMem() override { return 64u << 20; } HRESULT BeginScene() override { return S_OK; } HRESULT EndScene() override { return S_OK; } HRESULT SetTransform(D3DTRANSFORMSTATETYPE state, const D3DMATRIX* matrix) override { if (unsigned(state) < kTransforms && matrix) std::memcpy(m_transforms[state], matrix, sizeof(float) * 16); return S_OK; } HRESULT GetTransform(D3DTRANSFORMSTATETYPE state, D3DMATRIX* matrix) override { if (unsigned(state) < kTransforms) std::memcpy(matrix, m_transforms[state], sizeof(float) * 16); return S_OK; } HRESULT SetMaterial(const D3DMATERIAL8* material) override { m_material = *material; return S_OK; } HRESULT SetLight(DWORD index, const D3DLIGHT8* light) override { if (index < 8) m_lights[index] = *light; return S_OK; } HRESULT LightEnable(DWORD index, BOOL enable) override { if (index < 8) m_lightEnabled[index] = enable != FALSE; return S_OK; } HRESULT CreateTexture(UINT width, UINT height, UINT levels, DWORD usage, D3DFORMAT format, D3DPOOL pool, IDirect3DTexture8** out) override { if (!width || !height) return E_FAIL; auto* texture = new CpuTexture(); // Levels == 0 builds the full chain down to 1x1. for (UINT w = width, h = height; ; w = std::max(1u, w / 2), h = std::max(1u, h / 2)) { CpuTexture::Level level; level.desc = surface_desc(w, h, format, usage, pool); level.bytes.resize(level.desc.Size); texture->levels.push_back(std::move(level)); if ((levels && texture->levels.size() == levels) || (w == 1 && h == 1)) break; } *out = texture; return S_OK; } HRESULT CreateVertexBuffer(UINT length, DWORD, DWORD fvf, D3DPOOL, IDirect3DVertexBuffer8** out) override { auto* buffer = new DeviceVertexBuffer(); buffer->bytes.resize(length); buffer->fvf = fvf; *out = buffer; return S_OK; } HRESULT CreateIndexBuffer(UINT length, DWORD, D3DFORMAT format, D3DPOOL, IDirect3DIndexBuffer8** out) override { auto* buffer = new DeviceIndexBuffer(); buffer->bytes.resize(length); buffer->format = format; *out = buffer; return S_OK; } HRESULT CreateDepthStencilSurface(UINT width, UINT height, D3DFORMAT format, D3DMULTISAMPLE_TYPE, IDirect3DSurface8** out) override { auto* surface = new CpuSurface(); surface->desc = surface_desc(width, height, format, D3DUSAGE_DEPTHSTENCIL, D3DPOOL_DEFAULT); *out = surface; return S_OK; } HRESULT GetRenderTarget(IDirect3DSurface8** out) override { m_renderTarget->AddRef(); *out = m_renderTarget; return S_OK; } HRESULT GetDepthStencilSurface(IDirect3DSurface8** out) override { m_depthStencil->AddRef(); *out = m_depthStencil; return S_OK; } // Character shadow targets are rasterized below; other offscreen targets still need a // separate effect-specific implementation (snow blur is disabled by default). HRESULT SetRenderTarget(IDirect3DSurface8* target, IDirect3DSurface8* depth) override { if (target) { target->AddRef(); m_renderTarget->Release(); m_renderTarget = target; } if (depth) { depth->AddRef(); m_depthStencil->Release(); m_depthStencil = depth; } return S_OK; } HRESULT Clear(DWORD, const D3DRECT*, DWORD flags, D3DCOLOR color, float depth, DWORD) override { if (m_renderTarget == m_backBuffer) { // Kept in draw order: the renderer clears its colour/depth attachments at this point. Render3DDraw clear; clear.clear_flags = flags & (D3DCLEAR_TARGET | D3DCLEAR_ZBUFFER | D3DCLEAR_STENCIL); clear.clear_color = color; clear.clear_z = depth; if (clear.clear_flags) Render3DAdd(std::move(clear)); return S_OK; } auto* surface = static_cast(m_renderTarget); if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5) return S_OK; const size_t pixels = size_t(surface->desc.Width) * surface->desc.Height; if (flags & D3DCLEAR_TARGET) { const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 | (((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3)); auto& bytes = surface->parent->levels[0].bytes; bytes.resize(pixels * 2); for (size_t i = 0; i < pixels; ++i) std::memcpy(bytes.data() + i * 2, &rgb565, 2); ++surface->parent->revision; } if (flags & D3DCLEAR_ZBUFFER) m_offscreenDepth.assign(pixels, depth); return S_OK; } HRESULT SetRenderState(D3DRENDERSTATETYPE state, DWORD value) override { if (unsigned(state) < kRenderStates) m_renderStates[state] = value; return S_OK; } HRESULT SetTexture(DWORD stage, IDirect3DBaseTexture8* texture) override { if (stage < kStages) m_textures[stage] = texture; return S_OK; } HRESULT SetTextureStageState(DWORD stage, D3DTEXTURESTAGESTATETYPE type, DWORD value) override { if (stage < kStages && unsigned(type) < kStageStates) m_stageStates[stage][type] = value; return S_OK; } // PORT: fixed function only; CGraphicDevice's stream "shaders" are the FVFs they describe. HRESULT SetVertexShader(DWORD handle) override { m_fvf = handle; return S_OK; } HRESULT SetPixelShader(DWORD) override { return S_OK; } HRESULT SetVertexShaderConstant(DWORD, const void*, DWORD) override { return S_OK; } HRESULT SetPixelShaderConstant(DWORD, const void*, DWORD) override { return S_OK; } HRESULT SetStreamSource(UINT stream, IDirect3DVertexBuffer8* buffer, UINT stride) override { if (stream == 0) { m_stream = static_cast(buffer); m_streamStride = stride; } return S_OK; } HRESULT SetIndices(IDirect3DIndexBuffer8* indices, UINT baseVertex) override { m_indices = static_cast(indices); m_baseVertex = baseVertex; return S_OK; } HRESULT DrawPrimitive(D3DPRIMITIVETYPE type, UINT startVertex, UINT primitiveCount) override { if (!m_stream) return E_FAIL; const UINT count = index_count(type, primitiveCount); std::vector indices(count); for (UINT i = 0; i < count; ++i) indices[i] = startVertex + i; record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices, m_stream, nullptr, startVertex); return S_OK; } HRESULT DrawIndexedPrimitive(D3DPRIMITIVETYPE type, UINT, UINT, UINT startIndex, UINT primitiveCount) override { if (!m_stream || !m_indices) return E_FAIL; std::vector indices; if (!read_indices(m_indices->bytes.data(), m_indices->bytes.size(), m_indices->format, startIndex, index_count(type, primitiveCount), m_baseVertex, &indices)) return E_FAIL; record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices, m_stream, m_indices, startIndex); return S_OK; } HRESULT DrawPrimitiveUP(D3DPRIMITIVETYPE type, UINT primitiveCount, const void* vertices, UINT stride) override { const UINT count = index_count(type, primitiveCount); std::vector indices(count); for (UINT i = 0; i < count; ++i) indices[i] = i; record(type, primitiveCount, static_cast(vertices), size_t(count) * stride, stride, indices); return S_OK; } HRESULT DrawIndexedPrimitiveUP(D3DPRIMITIVETYPE type, UINT minVertex, UINT numVertices, UINT primitiveCount, const void* indexData, D3DFORMAT indexFormat, const void* vertices, UINT stride) override { const UINT count = index_count(type, primitiveCount); const size_t indexSize = indexFormat == D3DFMT_INDEX32 ? 4 : 2; std::vector indices; if (!read_indices(static_cast(indexData), count * indexSize, indexFormat, 0, count, 0, &indices)) return E_FAIL; record(type, primitiveCount, static_cast(vertices), size_t(minVertex + numVertices) * stride, stride, indices); return S_OK; } private: static constexpr unsigned kTransforms = 512; // D3DTS_WORLDMATRIX(0..255) = 256..511 static constexpr unsigned kRenderStates = 256; static constexpr unsigned kStages = 8; static constexpr unsigned kStageStates = 32; static UINT index_count(D3DPRIMITIVETYPE type, UINT primitives) { switch (type) { case D3DPT_POINTLIST: return primitives; case D3DPT_LINELIST: return primitives * 2; case D3DPT_LINESTRIP: return primitives + 1; case D3DPT_TRIANGLELIST: return primitives * 3; case D3DPT_TRIANGLESTRIP: case D3DPT_TRIANGLEFAN: return primitives + 2; default: return 0; } } static bool read_indices(const uint8_t* bytes, size_t size, D3DFORMAT format, UINT start, UINT count, UINT base, std::vector* out) { const size_t indexSize = format == D3DFMT_INDEX32 ? 4 : 2; if ((size_t(start) + count) * indexSize > size) return false; out->resize(count); for (UINT i = 0; i < count; ++i) { const uint8_t* p = bytes + (size_t(start) + i) * indexSize; std::uint32_t index = indexSize == 4 ? (p[0] | (p[1] << 8) | (p[2] << 16) | (std::uint32_t(p[3]) << 24)) : (p[0] | (p[1] << 8)); (*out)[i] = index + base; } return true; } static void transform_point(const float v[4], const float m[16], float out[4]) { for (int c = 0; c < 4; ++c) out[c] = v[0] * m[c] + v[1] * m[4 + c] + v[2] * m[8 + c] + v[3] * m[12 + c]; } static void multiply(const float a[16], const float b[16], float out[16]) { for (int r = 0; r < 4; ++r) transform_point(&a[r * 4], b, &out[r * 4]); } // The 40250 shadow pass renders a flat grey silhouette into an R5G6B5 target. // Rasterizing that small target here keeps the D3D render-target sequence and lets // the normal memory-texture path upload the result for terrain/object projection. void rasterize_shadow(const Render3DDraw& draw) { auto* surface = static_cast(m_renderTarget); if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5 || draw.lines) return; const int width = int(surface->desc.Width), height = int(surface->desc.Height); if (width <= 0 || height <= 0 || draw.indices.size() < 3) return; auto& bytes = surface->parent->levels[0].bytes; if (bytes.size() < size_t(width) * height * 2) return; if (m_offscreenDepth.size() != size_t(width) * height) m_offscreenDepth.assign(size_t(width) * height, 1.0f); float worldView[16], mvp[16]; multiply(draw.world, draw.view, worldView); multiply(worldView, draw.proj, mvp); struct Point { float x, y, z; bool valid; }; std::vector points(draw.positions.size() / 3); for (size_t i = 0; i < points.size(); ++i) { float p[4] = { draw.positions[i * 3], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f }; if (!draw.bone_matrices.empty() && draw.bone_weights.size() >= (i + 1) * 4 && draw.bone_indices.size() >= (i + 1) * 4) { float skinned[4] = {}; float total = 0.0f; for (int b = 0; b < 4; ++b) { const float weight = draw.bone_weights[i * 4 + b]; const size_t bone = draw.bone_indices[i * 4 + b]; if (weight <= 0.0f || (bone + 1) * 16 > draw.bone_matrices.size()) continue; float transformed[4]; transform_point(p, draw.bone_matrices.data() + bone * 16, transformed); for (int c = 0; c < 4; ++c) skinned[c] += transformed[c] * weight; total += weight; } if (total > 0.0f) { p[0] = skinned[0]; p[1] = skinned[1]; p[2] = skinned[2]; } } float clip[4]; transform_point(p, mvp, clip); const bool valid = clip[3] > 0.00001f; const float invW = valid ? 1.0f / clip[3] : 0.0f; points[i] = { draw.viewport[0] + (clip[0] * invW * 0.5f + 0.5f) * draw.viewport[2], draw.viewport[1] + (0.5f - clip[1] * invW * 0.5f) * draw.viewport[3], clip[2] * invW, valid }; } const std::uint32_t color = m_renderStates[D3DRS_TEXTUREFACTOR]; const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 | (((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3)); auto edge = [](const Point& a, const Point& b, float x, float y) { return (x - a.x) * (b.y - a.y) - (y - a.y) * (b.x - a.x); }; for (size_t i = 0; i + 2 < draw.indices.size(); i += 3) { if (draw.indices[i] >= points.size() || draw.indices[i + 1] >= points.size() || draw.indices[i + 2] >= points.size()) continue; const Point& a = points[draw.indices[i]]; const Point& b = points[draw.indices[i + 1]]; const Point& c = points[draw.indices[i + 2]]; if (!a.valid || !b.valid || !c.valid) continue; const float area = edge(a, b, c.x, c.y); if (std::fabs(area) < 0.00001f) continue; const int x0 = std::max(0, int(std::floor(std::min({a.x, b.x, c.x})))); const int y0 = std::max(0, int(std::floor(std::min({a.y, b.y, c.y})))); const int x1 = std::min(width - 1, int(std::ceil(std::max({a.x, b.x, c.x})))); const int y1 = std::min(height - 1, int(std::ceil(std::max({a.y, b.y, c.y})))); // D3D top-left fill rule: a pixel centre exactly on an edge is covered only when that edge // is a left edge (weight grows to the right) or a top edge (horizontal, weight grows downward). auto top_left = [area](const Point& p, const Point& q) { const float dx = (q.y - p.y) / area, dy = (p.x - q.x) / area; return dx > 0.0f || (dx == 0.0f && dy > 0.0f); }; const bool tlA = top_left(b, c), tlB = top_left(c, a), tlC = top_left(a, b); auto covered = [](float w, bool tl) { return w > 0.0f || (w == 0.0f && tl); }; for (int y = y0; y <= y1; ++y) for (int x = x0; x <= x1; ++x) { const float px = float(x) + 0.5f, py = float(y) + 0.5f; const float wa = edge(b, c, px, py) / area; const float wb = edge(c, a, px, py) / area; const float wc = edge(a, b, px, py) / area; if (!covered(wa, tlA) || !covered(wb, tlB) || !covered(wc, tlC)) continue; const float z = wa * a.z + wb * b.z + wc * c.z; const size_t pixel = size_t(y) * width + x; if (z < 0.0f || z > 1.0f || z > m_offscreenDepth[pixel]) continue; m_offscreenDepth[pixel] = z; std::memcpy(bytes.data() + pixel * 2, &rgb565, 2); } } } // Each two-triangle quad of a UI-space draw becomes an image command: its screen corners through // world * view * projection, its stage-0 texture (or D3DTA_TFACTOR colour when stage 0 selects it) // and, when stage 1 generates coordinates from the camera-space position through D3DTS_TEXTURE1 // (the minimap's circular filter), the mask's coordinates at the corners. void record_ui_quads(D3DPRIMITIVETYPE type, const uint8_t* vertices, size_t vertexBytes, UINT stride, const VertexLayout& layout, const std::vector& indices) { if (layout.uv0 < 0) return; unsigned width = 0, height = 0; UIRenderGetSize(&width, &height); float worldView[16], worldViewProj[16]; multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView); multiply(worldView, m_transforms[D3DTS_PROJECTION], worldViewProj); const DWORD* stage0 = m_stageStates[0]; const DWORD* stage1 = m_stageStates[1]; const bool factorColor = stage0[D3DTSS_COLOROP] == D3DTOP_SELECTARG1 && stage0[D3DTSS_COLORARG1] == D3DTA_TFACTOR; const std::uint32_t argb = factorColor ? (0xFF000000u | (m_renderStates[D3DRS_TEXTUREFACTOR] & 0x00FFFFFFu)) : 0xFFFFFFFFu; const bool masked = m_textures[1] && (stage1[D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == D3DTSS_TCI_CAMERASPACEPOSITION && stage1[D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2 && stage1[D3DTSS_COLOROP] == D3DTOP_MODULATE && stage1[D3DTSS_ALPHAOP] == D3DTOP_SELECTARG1 && stage1[D3DTSS_ALPHAARG1] == D3DTA_TEXTURE; const std::string texture = factorColor ? std::string() : UIRenderTextureNameFromHandle(m_textures[0]); const std::string mask = masked ? UIRenderTextureNameFromHandle(m_textures[1]) : std::string(); std::vector> quads; if (type == D3DPT_TRIANGLELIST) for (size_t i = 0; i + 6 <= indices.size(); i += 6) quads.push_back({ indices.begin() + i, indices.begin() + i + 6 }); else if (type == D3DPT_TRIANGLESTRIP && indices.size() == 4) quads.push_back(indices); for (const auto& quad : quads) { std::vector corners; for (std::uint32_t index : quad) if (std::find(corners.begin(), corners.end(), index) == corners.end()) corners.push_back(index); if (corners.size() != 4) continue; float sx[4], sy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; bool inside = true; for (int i = 0; i < 4; ++i) { const size_t offset = size_t(corners[i]) * stride; if (offset + stride > vertexBytes) { inside = false; break; } const uint8_t* vertex = vertices + offset; float position[4] = { 0, 0, 0, 1 }, clip[4], camera[4], coord[4]; std::memcpy(position, vertex, 12); std::memcpy(&u[i], vertex + layout.uv0, 4); std::memcpy(&v[i], vertex + layout.uv0 + 4, 4); transform_point(position, worldViewProj, clip); if (clip[3] == 0.0f) { inside = false; break; } sx[i] = (clip[0] / clip[3] * 0.5f + 0.5f) * float(width); sy[i] = (0.5f - clip[1] / clip[3] * 0.5f) * float(height); if (masked) { transform_point(position, worldView, camera); camera[3] = 1.0f; transform_point(camera, m_transforms[D3DTS_TEXTURE1], coord); mu[i] = coord[0]; mv[i] = coord[1]; } } if (!inside) continue; UIRenderCommand command{UIRenderCommand::Image, 0, 0, 0, 0, argb}; command.su = *std::min_element(u, u + 4); command.eu = *std::max_element(u, u + 4); command.sv = *std::min_element(v, v + 4); command.ev = *std::max_element(v, v + 4); // Corners in the command's TL, TR, BL, BR order, by texture coordinate. const float cu[4] = { command.su, command.eu, command.su, command.eu }; const float cv[4] = { command.sv, command.sv, command.ev, command.ev }; for (int c = 0; c < 4; ++c) { int best = 0; float bestDistance = INFINITY; for (int i = 0; i < 4; ++i) { const float distance = std::fabs(u[i] - cu[c]) + std::fabs(v[i] - cv[c]); if (distance < bestDistance) { bestDistance = distance; best = i; } } command.qx[c] = sx[best]; command.qy[c] = sy[best]; command.mu[c] = mu[best]; command.mv[c] = mv[best]; } command.x1 = *std::min_element(command.qx, command.qx + 4); command.y1 = *std::min_element(command.qy, command.qy + 4); command.x2 = *std::max_element(command.qx, command.qx + 4); command.y2 = *std::max_element(command.qy, command.qy + 4); command.text = texture; command.quad = true; command.mask = mask; UIRenderAdd(std::move(command)); } } void record(D3DPRIMITIVETYPE type, UINT primitives, const uint8_t* vertices, size_t vertexBytes, UINT stride, const std::vector& indices, const MtCpuVertexBuffer* source_vertex = nullptr, const MtCpuIndexBuffer* source_index = nullptr, UINT source_first_index = 0) { if (type == D3DPT_POINTLIST || indices.empty() || !vertices) return; VertexLayout layout = fvf_layout(m_fvf); if (!stride) stride = layout.stride; if (!stride || stride < (layout.rhw ? 16u : 12u)) return; // Elements past the stream's stride live in another stream (CreatePTStreamVertexShader). if (layout.normal + 12 > int(stride)) layout.normal = -1; if (layout.diffuse + 4 > int(stride)) layout.diffuse = -1; if (layout.uv0 + 8 > int(stride)) layout.uv0 = -1; if (layout.uv1 + 8 > int(stride)) layout.uv1 = -1; // An orthographic projection over untransformed vertices is a UI-space draw (CPythonMiniMap's // terrain tiles under CPythonGraphic::SetOrtho2D): it joins the UI command stream in order. if (m_renderTarget == m_backBuffer && !layout.rhw && m_transforms[D3DTS_PROJECTION][11] == 0.0f) { record_ui_quads(type, vertices, vertexBytes, stride, layout, indices); return; } Render3DDraw draw; std::memcpy(draw.world, m_transforms[256], sizeof(draw.world)); // D3DTS_WORLD std::memcpy(draw.view, m_transforms[D3DTS_VIEW], sizeof(draw.view)); std::memcpy(draw.proj, m_transforms[D3DTS_PROJECTION], sizeof(draw.proj)); draw.texture0 = UIRenderTextureNameFromHandle(m_textures[0]); draw.texture1 = UIRenderTextureNameFromHandle(m_textures[1]); draw.viewport[0] = float(m_viewport.X); draw.viewport[1] = float(m_viewport.Y); draw.viewport[2] = float(m_viewport.Width); draw.viewport[3] = float(m_viewport.Height); draw.viewport_z[0] = m_viewport.MinZ; draw.viewport_z[1] = m_viewport.MaxZ; draw.pretransformed = layout.rhw; draw.lines = type == D3DPT_LINELIST || type == D3DPT_LINESTRIP; // Rebase: copy only the vertices the indices reference. std::uint32_t lo = indices[0], hi = indices[0]; for (std::uint32_t index : indices) { lo = std::min(lo, index); hi = std::max(hi, index); } if ((size_t(hi) + 1) * stride > vertexBytes) return; const size_t count = size_t(hi - lo) + 1; GpuSkinSubrangeView skin_view; const bool gpu_skinned = source_vertex != nullptr && LookupGpuSkinSubrange(vertices, stride, lo, hi, &skin_view); if (source_vertex) { // The original buffers outlive individual draw calls. Keep their identity separate from // Unlock revisions so a static surface can keep its Godot mesh across frames. auto mix = [](std::uint64_t hash, std::uint64_t value) { return (hash ^ value) * 1099511628211ull; }; std::uint64_t key = 14695981039346656037ull; if (gpu_skinned) { key = mix(key, skin_view.source_mesh_key); key = mix(key, source_index ? source_index->id : 0); key = mix(key, source_first_index); key = mix(key, lo - skin_view.mesh_base_vertex); key = mix(key, hi - skin_view.mesh_base_vertex); key = mix(key, primitives); key = mix(key, type); key = mix(key, stride); key = mix(key, m_fvf); draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull; draw.geometry_revision = 1ull; } else { key = mix(key, source_vertex->id); key = mix(key, source_index ? source_index->id : 0); key = mix(key, source_first_index); key = mix(key, lo); key = mix(key, hi); key = mix(key, primitives); key = mix(key, type); key = mix(key, stride); key = mix(key, m_fvf); draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull; std::uint64_t revision = mix(14695981039346656037ull, source_vertex->revision); revision = mix(revision, source_index ? source_index->revision : 0); draw.geometry_revision = revision & 0x7FFFFFFFFFFFFFFFull; } } const std::uint64_t frame_id = UIRenderFrameId(); auto cached = draw.geometry_key ? m_geometry_cache.find(draw.geometry_key) : m_geometry_cache.end(); const bool reuse = cached != m_geometry_cache.end() && !cached->second.changing && cached->second.revision == draw.geometry_revision; if (reuse) { cached->second.last_used = frame_id; const Render3DDraw& prior = cached->second.geometry; draw.positions = prior.positions; draw.rhw = prior.rhw; draw.normals = prior.normals; draw.uv0 = prior.uv0; draw.uv1 = prior.uv1; draw.diffuse = prior.diffuse; draw.indices = prior.indices; draw.bone_indices = prior.bone_indices; draw.bone_weights = prior.bone_weights; } else { draw.positions.resize(count * 3); if (layout.rhw) draw.rhw.resize(count); if (layout.normal >= 0) draw.normals.resize(count * 3); if (layout.uv0 >= 0) draw.uv0.resize(count * 2); if (layout.uv1 >= 0) draw.uv1.resize(count * 2); if (layout.diffuse >= 0) draw.diffuse.resize(count); for (size_t i = 0; i < count; ++i) { const uint8_t* v = vertices + (lo + i) * stride; std::memcpy(&draw.positions[i * 3], v, 12); if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4); if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12); if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8); if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8); if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4); } if (gpu_skinned) { const std::size_t rel_lo = std::size_t(lo - skin_view.mesh_base_vertex); draw.bone_indices.resize(count * 4); draw.bone_weights.resize(count * 4); std::memcpy(draw.bone_indices.data(), skin_view.bone_indices + rel_lo * 4, count * 4 * sizeof(std::uint8_t)); std::memcpy(draw.bone_weights.data(), skin_view.bone_weights + rel_lo * 4, count * 4 * sizeof(float)); } // Strips and fans become lists; D3D flips the winding of every odd strip triangle. switch (type) { case D3DPT_TRIANGLESTRIP: for (UINT i = 0; i < primitives; ++i) { const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo; if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c }); else draw.indices.insert(draw.indices.end(), { a, b, c }); } break; case D3DPT_TRIANGLEFAN: for (UINT i = 0; i < primitives; ++i) draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo }); break; case D3DPT_LINESTRIP: for (UINT i = 0; i < primitives; ++i) draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo }); break; default: for (std::uint32_t index : indices) draw.indices.push_back(index - lo); break; } if (cached != m_geometry_cache.end()) { cached->second.last_used = frame_id; if (!cached->second.changing && cached->second.revision != draw.geometry_revision) { cached->second.changing = true; cached->second.geometry = Render3DDraw(); } } else if (draw.geometry_key && cached == m_geometry_cache.end()) m_geometry_cache.emplace(draw.geometry_key, GeometryCacheEntry{draw.geometry_revision, frame_id, false, draw}); } if (gpu_skinned && skin_view.bone_matrices && skin_view.bone_count > 0) draw.bone_matrices.assign(skin_view.bone_matrices, skin_view.bone_matrices + std::size_t(skin_view.bone_count) * 16); if (frame_id >= m_cache_pruned_frame + 120) { m_cache_pruned_frame = frame_id; for (auto it = m_geometry_cache.begin(); it != m_geometry_cache.end();) it = frame_id - it->second.last_used > 120 ? m_geometry_cache.erase(it) : std::next(it); } draw.alpha_blend = m_renderStates[D3DRS_ALPHABLENDENABLE]; draw.src_blend = m_renderStates[D3DRS_SRCBLEND]; draw.dest_blend = m_renderStates[D3DRS_DESTBLEND]; draw.alpha_test = m_renderStates[D3DRS_ALPHATESTENABLE]; draw.alpha_ref = m_renderStates[D3DRS_ALPHAREF]; draw.alpha_func = m_renderStates[D3DRS_ALPHAFUNC]; draw.cull_mode = m_renderStates[D3DRS_CULLMODE]; draw.z_enable = m_renderStates[D3DRS_ZENABLE]; draw.z_write = m_renderStates[D3DRS_ZWRITEENABLE]; draw.z_func = m_renderStates[D3DRS_ZFUNC]; draw.lighting = m_renderStates[D3DRS_LIGHTING]; draw.texture_factor = m_renderStates[D3DRS_TEXTUREFACTOR]; draw.fog_enable = m_renderStates[D3DRS_FOGENABLE]; draw.fog_color = m_renderStates[D3DRS_FOGCOLOR]; draw.fog_vertex_mode = m_renderStates[D3DRS_FOGVERTEXMODE]; draw.fog_table_mode = m_renderStates[D3DRS_FOGTABLEMODE]; draw.fog_range_enable = m_renderStates[D3DRS_RANGEFOGENABLE]; std::memcpy(&draw.fog_start, &m_renderStates[D3DRS_FOGSTART], sizeof(float)); std::memcpy(&draw.fog_end, &m_renderStates[D3DRS_FOGEND], sizeof(float)); std::memcpy(&draw.fog_density, &m_renderStates[D3DRS_FOGDENSITY], sizeof(float)); draw.ambient = m_renderStates[D3DRS_AMBIENT]; draw.diffuse_material_source = m_renderStates[D3DRS_DIFFUSEMATERIALSOURCE]; draw.ambient_material_source = m_renderStates[D3DRS_AMBIENTMATERIALSOURCE]; draw.emissive_material_source = m_renderStates[D3DRS_EMISSIVEMATERIALSOURCE]; draw.color_vertex = m_renderStates[D3DRS_COLORVERTEX]; draw.normalize_normals = m_renderStates[D3DRS_NORMALIZENORMALS]; draw.local_viewer = m_renderStates[D3DRS_LOCALVIEWER]; for (int stage = 0; stage < 2; ++stage) { draw.color_op[stage] = m_stageStates[stage][D3DTSS_COLOROP]; draw.color_arg1[stage] = m_stageStates[stage][D3DTSS_COLORARG1]; draw.color_arg2[stage] = m_stageStates[stage][D3DTSS_COLORARG2]; draw.alpha_op[stage] = m_stageStates[stage][D3DTSS_ALPHAOP]; draw.alpha_arg1[stage] = m_stageStates[stage][D3DTSS_ALPHAARG1]; draw.alpha_arg2[stage] = m_stageStates[stage][D3DTSS_ALPHAARG2]; draw.address_u[stage] = m_stageStates[stage][D3DTSS_ADDRESSU]; draw.address_v[stage] = m_stageStates[stage][D3DTSS_ADDRESSV]; draw.min_filter[stage] = m_stageStates[stage][D3DTSS_MINFILTER]; draw.mag_filter[stage] = m_stageStates[stage][D3DTSS_MAGFILTER]; draw.mip_filter[stage] = m_stageStates[stage][D3DTSS_MIPFILTER]; draw.border_color[stage] = m_stageStates[stage][D3DTSS_BORDERCOLOR]; draw.texcoord_index[stage] = m_stageStates[stage][D3DTSS_TEXCOORDINDEX]; draw.texture_transform_flags[stage] = m_stageStates[stage][D3DTSS_TEXTURETRANSFORMFLAGS]; std::memcpy(draw.texture_matrix[stage], m_transforms[D3DTS_TEXTURE0 + stage], sizeof(draw.texture_matrix[stage])); } copy_color(draw.material_diffuse, m_material.Diffuse); copy_color(draw.material_ambient, m_material.Ambient); copy_color(draw.material_emissive, m_material.Emissive); if (m_lightEnabled[0] && (m_lights[0].Type == D3DLIGHT_DIRECTIONAL || m_lights[0].Type == D3DLIGHT_SPOT)) { draw.light0 = true; draw.light0_direction[0] = m_lights[0].Direction.x; draw.light0_direction[1] = m_lights[0].Direction.y; draw.light0_direction[2] = m_lights[0].Direction.z; copy_color(draw.light0_diffuse, m_lights[0].Diffuse); copy_color(draw.light0_ambient, m_lights[0].Ambient); } for (int i = 0; i < 8; ++i) { if (!m_lightEnabled[i]) continue; const D3DLIGHT8& source = m_lights[i]; auto& light = draw.lights[i]; light.type = source.Type; light.position[0] = source.Position.x; light.position[1] = source.Position.y; light.position[2] = source.Position.z; light.direction[0] = source.Direction.x; light.direction[1] = source.Direction.y; light.direction[2] = source.Direction.z; copy_color(light.diffuse, source.Diffuse); copy_color(light.ambient, source.Ambient); light.attenuation[0] = source.Attenuation0; light.attenuation[1] = source.Attenuation1; light.attenuation[2] = source.Attenuation2; light.range = source.Range; light.theta = source.Theta; light.phi = source.Phi; light.falloff = source.Falloff; } if (m_renderTarget == m_backBuffer) Render3DAdd(std::move(draw)); else rasterize_shadow(draw); } ULONG m_refs = 1; int m_width, m_height; D3DVIEWPORT8 m_viewport = {}; float m_transforms[kTransforms][16]; DWORD m_renderStates[kRenderStates] = {}; DWORD m_stageStates[kStages][kStageStates] = {}; IDirect3DBaseTexture8* m_textures[kStages] = {}; D3DMATERIAL8 m_material; D3DLIGHT8 m_lights[8]; bool m_lightEnabled[8] = {}; IDirect3DSurface8* m_backBuffer = nullptr; IDirect3DSurface8* m_depthBuffer = nullptr; IDirect3DSurface8* m_renderTarget = nullptr; IDirect3DSurface8* m_depthStencil = nullptr; DWORD m_fvf = 0; MtCpuVertexBuffer* m_stream = nullptr; UINT m_streamStride = 0; MtCpuIndexBuffer* m_indices = nullptr; UINT m_baseVertex = 0; struct GeometryCacheEntry { std::uint64_t revision; std::uint64_t last_used; bool changing; Render3DDraw geometry; }; std::unordered_map m_geometry_cache; std::uint64_t m_cache_pruned_frame = 0; std::vector m_offscreenDepth; }; } std::string MtCpuTextureNameFromHandle(const IDirect3DBaseTexture8* handle) { if (!handle) return {}; std::lock_guard lock(g_cpu_tex_mutex); for (const auto& entry : g_cpu_textures) { if (entry.second == handle && !entry.second->levels.empty()) return "mem:cpu_" + std::to_string(entry.first) + "@" + std::to_string(entry.second->revision); } return {}; } bool MtCpuMemoryTexture(const std::string& name, UIMemoryTexture* out) { if (name.compare(0, 8, "mem:cpu_") != 0 || !out) return false; const unsigned id = unsigned(std::strtoul(name.c_str() + 8, nullptr, 10)); std::lock_guard lock(g_cpu_tex_mutex); auto it = g_cpu_textures.find(id); if (it == g_cpu_textures.end() || it->second->levels.empty()) return false; const auto& lv0 = it->second->levels[0]; const UINT w = lv0.desc.Width; const UINT h = lv0.desc.Height; if (!w || !h) return false; out->width = static_cast(w); out->height = static_cast(h); out->revision = it->second->revision; out->argb.resize(std::size_t(w) * std::size_t(h)); const uint8_t* bytes = lv0.bytes.data(); const int bpp = format_bytes(lv0.desc.Format); for (std::size_t i = 0; i < out->argb.size(); ++i) { if (bpp == 4) { std::uint32_t px = 0; std::memcpy(&px, bytes + i * 4, 4); // CTerrain::PutImage32 packs alpha into bits 31..24 with RGB == 0. // Set RGB to white so stage-1 modulation preserves the base texture's RGB. if ((px & 0x00FFFFFFu) == 0) px |= 0x00FFFFFFu; out->argb[i] = px; } else if (lv0.desc.Format == D3DFMT_R5G6B5) { std::uint16_t word = 0; std::memcpy(&word, bytes + i * 2, 2); const std::uint32_t r = ((word >> 11) & 31u) * 255u / 31u; const std::uint32_t g = ((word >> 5) & 63u) * 255u / 63u; const std::uint32_t b = (word & 31u) * 255u / 31u; out->argb[i] = 0xFF000000u | (r << 16) | (g << 8) | b; } else if (bpp == 2) { std::uint16_t word = 0; std::memcpy(&word, bytes + i * 2, 2); // CTerrain::PutImage16 writes `src[x] << 8`, storing the full 8-bit alpha in the high byte. const std::uint32_t a = (word >> 8) & 0xFFu; out->argb[i] = (a << 24) | 0x00FFFFFFu; } else { const std::uint32_t a = bytes[i]; out->argb[i] = (a << 24) | 0x00FFFFFFu; } } return true; } IDirect3DDevice8* MtCreateRecordingDevice(int width, int height) { return new RecordingDevice(width, height); } void SetNativeTerrainRenderEnabled(bool enabled) { g_native_terrain_override = enabled ? 1 : 0; } bool IsNativeTerrainRenderEnabled() { if (g_native_terrain_override >= 0) return g_native_terrain_override != 0; static const bool env_on = [] { const char* v = std::getenv("MT_NATIVE_TERRAIN"); return v && (*v == '1' || *v == 't' || *v == 'T' || *v == 'y' || *v == 'Y'); }(); return env_on; } void Render3DBeginFrame() { std::lock_guard lock(g_draws_mutex); g_draws.clear(); } void Render3DAdd(Render3DDraw draw) { std::lock_guard lock(g_draws_mutex); g_draws.push_back(std::move(draw)); } const std::vector& Render3DDraws() { return g_draws; } // D3D8 fixed-function texture coordinate processing for one stage (the native renderer's // native.vert applies the same rules on the GPU): // - D3DTSS_TEXCOORDINDEX low word picks the vertex set, a 2D set enters the transform as (u, v, 1, 0) // (so a D3DTS_TEXTUREn translation lives in _31/_32, as CSkyBox::RenderCloud writes it); // - D3DTSS_TCI_CAMERASPACENORMAL/POSITION/REFLECTIONVECTOR generate (x, y, z, 1) in camera space; // - D3DTSS_TEXTURETRANSFORMFLAGS != DISABLE multiplies by D3DTS_TEXTUREn (row vector), and // D3DTTFF_PROJECTED divides by the last counted element. void Render3DStageTexcoords(const Render3DDraw& draw, int stage, std::vector& out) { const size_t count = draw.positions.size() / 3; out.assign(count * 2, 0.0f); if (stage < 0 || stage > 1) return; const std::uint32_t tci = draw.texcoord_index[stage]; const std::uint32_t gen = tci & 0xFFFF0000u; const std::uint32_t set = tci & 0xFFFFu; const std::vector* uv = set == 0 ? &draw.uv0 : (set == 1 ? &draw.uv1 : nullptr); const std::uint32_t flags = draw.texture_transform_flags[stage]; const std::uint32_t elements = flags & 0xFFu; const bool projected = (flags & D3DTTFF_PROJECTED) != 0; float worldView[16] = {}; for (int row = 0; row < 4; ++row) for (int col = 0; col < 4; ++col) for (int k = 0; k < 4; ++k) worldView[row * 4 + col] += draw.world[row * 4 + k] * draw.view[k * 4 + col]; // Normals go through the inverse transpose of world * view (cofactors / determinant). const float* w = worldView; float normalMatrix[9] = { w[5] * w[10] - w[6] * w[9], -(w[4] * w[10] - w[6] * w[8]), w[4] * w[9] - w[5] * w[8], -(w[1] * w[10] - w[2] * w[9]), w[0] * w[10] - w[2] * w[8], -(w[0] * w[9] - w[1] * w[8]), w[1] * w[6] - w[2] * w[5], -(w[0] * w[6] - w[2] * w[4]), w[0] * w[5] - w[1] * w[4] }; const float det = w[0] * normalMatrix[0] + w[1] * normalMatrix[1] + w[2] * normalMatrix[2]; if (det != 0.0f) for (float& c : normalMatrix) c /= det; const float* m = draw.texture_matrix[stage]; for (size_t i = 0; i < count; ++i) { float in[4] = { 0, 0, 1, 0 }; if (gen == 0 || draw.pretransformed) { if (uv && uv->size() >= (i + 1) * 2) { in[0] = (*uv)[i * 2 + 0]; in[1] = (*uv)[i * 2 + 1]; } } else { const float* p = &draw.positions[i * 3]; float eye[3], n[3] = { 0, 0, 0 }; for (int c = 0; c < 3; ++c) eye[c] = p[0] * worldView[c] + p[1] * worldView[4 + c] + p[2] * worldView[8 + c] + worldView[12 + c]; if (draw.normals.size() >= (i + 1) * 3) { const float* src = &draw.normals[i * 3]; for (int c = 0; c < 3; ++c) n[c] = src[0] * normalMatrix[c] + src[1] * normalMatrix[3 + c] + src[2] * normalMatrix[6 + c]; if (draw.normalize_normals) { const float len = std::sqrt(n[0] * n[0] + n[1] * n[1] + n[2] * n[2]); if (len > 0.0f) for (float& c : n) c /= len; } } float v[3] = { 0, 0, 0 }; if (gen == D3DTSS_TCI_CAMERASPACENORMAL) std::memcpy(v, n, sizeof(v)); else if (gen == D3DTSS_TCI_CAMERASPACEPOSITION) std::memcpy(v, eye, sizeof(v)); else if (gen == D3DTSS_TCI_CAMERASPACEREFLECTIONVECTOR) { float e[3] = { 0, 0, -1 }; if (draw.local_viewer) { const float len = std::sqrt(eye[0] * eye[0] + eye[1] * eye[1] + eye[2] * eye[2]); for (int c = 0; c < 3; ++c) e[c] = len > 0.0f ? -eye[c] / len : 0.0f; } const float en = e[0] * n[0] + e[1] * n[1] + e[2] * n[2]; for (int c = 0; c < 3; ++c) v[c] = 2.0f * en * n[c] - e[c]; } in[0] = v[0]; in[1] = v[1]; in[2] = v[2]; in[3] = 1.0f; } float r[4] = { in[0], in[1], in[2], in[3] }; // Pre-transformed (XYZRHW) vertices pass their coordinates through untransformed. if (elements != D3DTTFF_DISABLE && !draw.pretransformed) for (int c = 0; c < 4; ++c) r[c] = in[0] * m[c] + in[1] * m[4 + c] + in[2] * m[8 + c] + in[3] * m[12 + c]; if (projected && !draw.pretransformed && elements >= 2 && elements <= 4 && r[elements - 1] != 0.0f) { r[0] /= r[elements - 1]; r[1] /= r[elements - 1]; } out[i * 2 + 0] = r[0]; out[i * 2 + 1] = r[1]; } }