2V2-e.1: 3D draw-command channel — recording D3D8 device, StateManager, GrpBase, camera

EterLib/StateManager.cpp and GrpBase.cpp ported verbatim. CGraphicDevice::Create builds a platform
IDirect3DDevice8 that keeps the device state and records every Draw* call as a Render3DDraw
(vertices decoded by FVF, strips/fans expanded, textures, world/view/proj, blend/depth/light state).
The 40250 camera (__UpdateCamera, SetCenterPosition, ...) and the Process order
(__UpdateCamera -> OnCameraUpdate -> OnUIUpdate -> Begin/SetInterfaceRenderState/OnUIRender/End)
are in place; CScreen Begin/End, CPythonGraphic render states and CCullingManager::Process matrix
updates are verbatim. port.login_flow asserts a textured, lit character draw under the RH
perspective with every vertex on screen; offline and live pass.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-23 20:15:54 +09:00
co-authored by Claude Opus 5.5
parent 668f48d6ec
commit 5bfa13c5cc
27 changed files with 3616 additions and 777 deletions
@@ -0,0 +1,385 @@
#include "RecordingDevice.h"
#include "RenderCommands3D.h"
#include "UIRenderCommands.h"
#include "CpuBuffer.h"
#include <cstring>
#include <mutex>
namespace {
std::mutex g_draws_mutex;
std::vector<Render3DDraw> g_draws;
struct VertexLayout {
unsigned stride = 0;
bool rhw = false;
int normal = -1, diffuse = -1, uv0 = -1, uv1 = -1;
};
// The element offsets of a fixed-function FVF (D3DXGetFVFVertexSize's order).
VertexLayout fvf_layout(DWORD fvf)
{
VertexLayout layout;
unsigned offset = 0;
switch (fvf & D3DFVF_POSITION_MASK)
{
case D3DFVF_XYZ: offset = 12; break;
case D3DFVF_XYZRHW: offset = 16; layout.rhw = true; break;
case D3DFVF_XYZB1: offset = 16; break;
case D3DFVF_XYZB2: offset = 20; break;
case D3DFVF_XYZB3: offset = 24; break;
case D3DFVF_XYZB4: offset = 28; break;
case D3DFVF_XYZB5: offset = 32; break;
}
if (fvf & D3DFVF_NORMAL) { layout.normal = offset; offset += 12; }
if (fvf & D3DFVF_PSIZE) offset += 4;
if (fvf & D3DFVF_DIFFUSE) { layout.diffuse = offset; offset += 4; }
if (fvf & D3DFVF_SPECULAR) offset += 4;
const unsigned texCount = (fvf & D3DFVF_TEXCOUNT_MASK) >> D3DFVF_TEXCOUNT_SHIFT;
for (unsigned i = 0; i < texCount; ++i)
{
static const unsigned coordSize[4] = { 8, 12, 16, 4 };
if (i == 0) layout.uv0 = offset;
if (i == 1) layout.uv1 = offset;
offset += coordSize[(fvf >> (16 + i * 2)) & 3];
}
layout.stride = offset;
return layout;
}
void copy_color(float out[4], const D3DCOLORVALUE& c)
{
out[0] = c.r; out[1] = c.g; out[2] = c.b; out[3] = c.a;
}
class RecordingDevice final : public IDirect3DDevice8
{
public:
RecordingDevice(int width, int height) : m_width(width), m_height(height)
{
static const float identity[16] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 };
for (auto& matrix : m_transforms)
std::memcpy(matrix, identity, sizeof(identity));
std::memset(&m_material, 0, sizeof(m_material));
m_material.Diffuse = { 1, 1, 1, 1 };
std::memset(m_lights, 0, sizeof(m_lights));
}
ULONG AddRef() override { return ++m_refs; }
ULONG Release() override
{
const ULONG refs = --m_refs;
if (!refs)
delete this;
return refs;
}
// PORT: a D3D8 HAL on a T&L card (D3DDEVCAPS_HWTRANSFORMANDLIGHT), 8 lights, 4 texture stages.
HRESULT GetDeviceCaps(D3DCAPS8* caps) override
{
std::memset(caps, 0, sizeof(*caps));
caps->DevCaps = D3DDEVCAPS_HWTRANSFORMANDLIGHT;
caps->MaxActiveLights = 8;
caps->MaxTextureBlendStages = 4;
caps->MaxSimultaneousTextures = 4;
caps->MaxTextureWidth = caps->MaxTextureHeight = 4096;
caps->MaxAnisotropy = 16;
caps->MaxPrimitiveCount = 0xFFFFF;
caps->MaxVertexIndex = 0xFFFFF;
caps->MaxStreams = 8;
caps->MaxStreamStride = 255;
caps->TextureAddressCaps = D3DPTADDRESSCAPS_BORDER | D3DPTADDRESSCAPS_CLAMP | D3DPTADDRESSCAPS_WRAP | D3DPTADDRESSCAPS_MIRROR;
return S_OK;
}
HRESULT GetViewport(D3DVIEWPORT8* viewport) override
{
*viewport = { 0, 0, DWORD(m_width), DWORD(m_height), 0.0f, 1.0f };
return S_OK;
}
// PORT: CPU memory has no texture budget; report the 64MB of a mid-range 2004 card.
UINT GetAvailableTextureMem() override { return 64u << 20; }
HRESULT BeginScene() override { return S_OK; }
HRESULT EndScene() override { return S_OK; }
HRESULT SetTransform(D3DTRANSFORMSTATETYPE state, const D3DMATRIX* matrix) override
{
if (unsigned(state) < kTransforms && matrix)
std::memcpy(m_transforms[state], matrix, sizeof(float) * 16);
return S_OK;
}
HRESULT GetTransform(D3DTRANSFORMSTATETYPE state, D3DMATRIX* matrix) override
{
if (unsigned(state) < kTransforms)
std::memcpy(matrix, m_transforms[state], sizeof(float) * 16);
return S_OK;
}
HRESULT SetMaterial(const D3DMATERIAL8* material) override { m_material = *material; return S_OK; }
HRESULT SetLight(DWORD index, const D3DLIGHT8* light) override
{
if (index < 8)
m_lights[index] = *light;
return S_OK;
}
HRESULT SetRenderState(D3DRENDERSTATETYPE state, DWORD value) override
{
if (unsigned(state) < kRenderStates)
m_renderStates[state] = value;
return S_OK;
}
HRESULT SetTexture(DWORD stage, IDirect3DBaseTexture8* texture) override
{
if (stage < kStages)
m_textures[stage] = texture;
return S_OK;
}
HRESULT SetTextureStageState(DWORD stage, D3DTEXTURESTAGESTATETYPE type, DWORD value) override
{
if (stage < kStages && unsigned(type) < kStageStates)
m_stageStates[stage][type] = value;
return S_OK;
}
// PORT: fixed function only; CGraphicDevice's stream "shaders" are the FVFs they describe.
HRESULT SetVertexShader(DWORD handle) override { m_fvf = handle; return S_OK; }
HRESULT SetPixelShader(DWORD) override { return S_OK; }
HRESULT SetVertexShaderConstant(DWORD, const void*, DWORD) override { return S_OK; }
HRESULT SetPixelShaderConstant(DWORD, const void*, DWORD) override { return S_OK; }
HRESULT SetStreamSource(UINT stream, IDirect3DVertexBuffer8* buffer, UINT stride) override
{
if (stream == 0)
{
m_stream = static_cast<MtCpuVertexBuffer*>(buffer);
m_streamStride = stride;
}
return S_OK;
}
HRESULT SetIndices(IDirect3DIndexBuffer8* indices, UINT baseVertex) override
{
m_indices = static_cast<MtCpuIndexBuffer*>(indices);
m_baseVertex = baseVertex;
return S_OK;
}
HRESULT DrawPrimitive(D3DPRIMITIVETYPE type, UINT startVertex, UINT primitiveCount) override
{
if (!m_stream)
return E_FAIL;
const UINT count = index_count(type, primitiveCount);
std::vector<std::uint32_t> indices(count);
for (UINT i = 0; i < count; ++i)
indices[i] = startVertex + i;
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices);
return S_OK;
}
HRESULT DrawIndexedPrimitive(D3DPRIMITIVETYPE type, UINT, UINT, UINT startIndex, UINT primitiveCount) override
{
if (!m_stream || !m_indices)
return E_FAIL;
std::vector<std::uint32_t> indices;
if (!read_indices(m_indices->bytes.data(), m_indices->bytes.size(), m_indices->format, startIndex,
index_count(type, primitiveCount), m_baseVertex, &indices))
return E_FAIL;
record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices);
return S_OK;
}
HRESULT DrawPrimitiveUP(D3DPRIMITIVETYPE type, UINT primitiveCount, const void* vertices, UINT stride) override
{
const UINT count = index_count(type, primitiveCount);
std::vector<std::uint32_t> indices(count);
for (UINT i = 0; i < count; ++i)
indices[i] = i;
record(type, primitiveCount, static_cast<const uint8_t*>(vertices), size_t(count) * stride, stride, indices);
return S_OK;
}
HRESULT DrawIndexedPrimitiveUP(D3DPRIMITIVETYPE type, UINT minVertex, UINT numVertices, UINT primitiveCount,
const void* indexData, D3DFORMAT indexFormat, const void* vertices, UINT stride) override
{
const UINT count = index_count(type, primitiveCount);
const size_t indexSize = indexFormat == D3DFMT_INDEX32 ? 4 : 2;
std::vector<std::uint32_t> indices;
if (!read_indices(static_cast<const uint8_t*>(indexData), count * indexSize, indexFormat, 0, count, 0, &indices))
return E_FAIL;
record(type, primitiveCount, static_cast<const uint8_t*>(vertices), size_t(minVertex + numVertices) * stride, stride, indices);
return S_OK;
}
private:
static constexpr unsigned kTransforms = 512; // D3DTS_WORLDMATRIX(0..255) = 256..511
static constexpr unsigned kRenderStates = 256;
static constexpr unsigned kStages = 8;
static constexpr unsigned kStageStates = 32;
static UINT index_count(D3DPRIMITIVETYPE type, UINT primitives)
{
switch (type)
{
case D3DPT_POINTLIST: return primitives;
case D3DPT_LINELIST: return primitives * 2;
case D3DPT_LINESTRIP: return primitives + 1;
case D3DPT_TRIANGLELIST: return primitives * 3;
case D3DPT_TRIANGLESTRIP: case D3DPT_TRIANGLEFAN: return primitives + 2;
default: return 0;
}
}
static bool read_indices(const uint8_t* bytes, size_t size, D3DFORMAT format, UINT start, UINT count, UINT base,
std::vector<std::uint32_t>* out)
{
const size_t indexSize = format == D3DFMT_INDEX32 ? 4 : 2;
if ((size_t(start) + count) * indexSize > size)
return false;
out->resize(count);
for (UINT i = 0; i < count; ++i)
{
const uint8_t* p = bytes + (size_t(start) + i) * indexSize;
std::uint32_t index = indexSize == 4 ? (p[0] | (p[1] << 8) | (p[2] << 16) | (std::uint32_t(p[3]) << 24)) : (p[0] | (p[1] << 8));
(*out)[i] = index + base;
}
return true;
}
void record(D3DPRIMITIVETYPE type, UINT primitives, const uint8_t* vertices, size_t vertexBytes, UINT stride,
const std::vector<std::uint32_t>& indices)
{
if (type == D3DPT_POINTLIST || indices.empty() || !vertices)
return;
VertexLayout layout = fvf_layout(m_fvf);
if (!stride)
stride = layout.stride;
if (!stride || stride < (layout.rhw ? 16u : 12u))
return;
// Elements past the stream's stride live in another stream (CreatePTStreamVertexShader).
if (layout.normal + 12 > int(stride)) layout.normal = -1;
if (layout.diffuse + 4 > int(stride)) layout.diffuse = -1;
if (layout.uv0 + 8 > int(stride)) layout.uv0 = -1;
if (layout.uv1 + 8 > int(stride)) layout.uv1 = -1;
Render3DDraw draw;
std::memcpy(draw.world, m_transforms[256], sizeof(draw.world)); // D3DTS_WORLD
std::memcpy(draw.view, m_transforms[D3DTS_VIEW], sizeof(draw.view));
std::memcpy(draw.proj, m_transforms[D3DTS_PROJECTION], sizeof(draw.proj));
draw.texture0 = UIRenderTextureNameFromHandle(m_textures[0]);
draw.texture1 = UIRenderTextureNameFromHandle(m_textures[1]);
draw.pretransformed = layout.rhw;
draw.lines = type == D3DPT_LINELIST || type == D3DPT_LINESTRIP;
// Rebase: copy only the vertices the indices reference.
std::uint32_t lo = indices[0], hi = indices[0];
for (std::uint32_t index : indices)
{
lo = std::min(lo, index);
hi = std::max(hi, index);
}
if ((size_t(hi) + 1) * stride > vertexBytes)
return;
const size_t count = size_t(hi - lo) + 1;
draw.positions.resize(count * 3);
if (layout.rhw) draw.rhw.resize(count);
if (layout.normal >= 0) draw.normals.resize(count * 3);
if (layout.uv0 >= 0) draw.uv0.resize(count * 2);
if (layout.uv1 >= 0) draw.uv1.resize(count * 2);
if (layout.diffuse >= 0) draw.diffuse.resize(count);
for (size_t i = 0; i < count; ++i)
{
const uint8_t* v = vertices + (lo + i) * stride;
std::memcpy(&draw.positions[i * 3], v, 12);
if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4);
if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12);
if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8);
if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8);
if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4);
}
// Strips and fans become lists; D3D flips the winding of every odd strip triangle.
switch (type)
{
case D3DPT_TRIANGLESTRIP:
for (UINT i = 0; i < primitives; ++i)
{
const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo;
if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c });
else draw.indices.insert(draw.indices.end(), { a, b, c });
}
break;
case D3DPT_TRIANGLEFAN:
for (UINT i = 0; i < primitives; ++i)
draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo });
break;
case D3DPT_LINESTRIP:
for (UINT i = 0; i < primitives; ++i)
draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo });
break;
default:
for (std::uint32_t index : indices)
draw.indices.push_back(index - lo);
break;
}
draw.alpha_blend = m_renderStates[D3DRS_ALPHABLENDENABLE];
draw.src_blend = m_renderStates[D3DRS_SRCBLEND];
draw.dest_blend = m_renderStates[D3DRS_DESTBLEND];
draw.alpha_test = m_renderStates[D3DRS_ALPHATESTENABLE];
draw.alpha_ref = m_renderStates[D3DRS_ALPHAREF];
draw.alpha_func = m_renderStates[D3DRS_ALPHAFUNC];
draw.cull_mode = m_renderStates[D3DRS_CULLMODE];
draw.z_enable = m_renderStates[D3DRS_ZENABLE];
draw.z_write = m_renderStates[D3DRS_ZWRITEENABLE];
draw.z_func = m_renderStates[D3DRS_ZFUNC];
draw.lighting = m_renderStates[D3DRS_LIGHTING];
draw.texture_factor = m_renderStates[D3DRS_TEXTUREFACTOR];
draw.fog_enable = m_renderStates[D3DRS_FOGENABLE];
draw.ambient = m_renderStates[D3DRS_AMBIENT];
for (int stage = 0; stage < 2; ++stage)
{
draw.color_op[stage] = m_stageStates[stage][D3DTSS_COLOROP];
draw.color_arg1[stage] = m_stageStates[stage][D3DTSS_COLORARG1];
draw.color_arg2[stage] = m_stageStates[stage][D3DTSS_COLORARG2];
draw.alpha_op[stage] = m_stageStates[stage][D3DTSS_ALPHAOP];
draw.alpha_arg1[stage] = m_stageStates[stage][D3DTSS_ALPHAARG1];
draw.alpha_arg2[stage] = m_stageStates[stage][D3DTSS_ALPHAARG2];
}
copy_color(draw.material_diffuse, m_material.Diffuse);
copy_color(draw.material_ambient, m_material.Ambient);
copy_color(draw.material_emissive, m_material.Emissive);
if (m_lights[0].Type == D3DLIGHT_DIRECTIONAL)
{
draw.light0 = true;
draw.light0_direction[0] = m_lights[0].Direction.x;
draw.light0_direction[1] = m_lights[0].Direction.y;
draw.light0_direction[2] = m_lights[0].Direction.z;
copy_color(draw.light0_diffuse, m_lights[0].Diffuse);
copy_color(draw.light0_ambient, m_lights[0].Ambient);
}
Render3DAdd(std::move(draw));
}
ULONG m_refs = 1;
int m_width, m_height;
float m_transforms[kTransforms][16];
DWORD m_renderStates[kRenderStates] = {};
DWORD m_stageStates[kStages][kStageStates] = {};
IDirect3DBaseTexture8* m_textures[kStages] = {};
D3DMATERIAL8 m_material;
D3DLIGHT8 m_lights[8];
DWORD m_fvf = 0;
MtCpuVertexBuffer* m_stream = nullptr;
UINT m_streamStride = 0;
MtCpuIndexBuffer* m_indices = nullptr;
UINT m_baseVertex = 0;
};
}
IDirect3DDevice8* MtCreateRecordingDevice(int width, int height) { return new RecordingDevice(width, height); }
void Render3DBeginFrame()
{
std::lock_guard<std::mutex> lock(g_draws_mutex);
g_draws.clear();
}
void Render3DAdd(Render3DDraw draw)
{
std::lock_guard<std::mutex> lock(g_draws_mutex);
g_draws.push_back(std::move(draw));
}
const std::vector<Render3DDraw>& Render3DDraws() { return g_draws; }