diff --git a/CMakeLists.txt b/CMakeLists.txt index 5c34c8db..f35b1112 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -23,6 +23,7 @@ option(MTGODOT_BUILD_EXTENSION "Build the Godot GDExtension (pulls in godot-cpp) option(MTGODOT_BUILD_TOOLS "Build libgr2 validation tools (gr2dump/gr2fuzz/oracle_diff)" OFF) option(MTGODOT_BUILD_NET_TOOLS "Build host-only protocol probes without Godot" OFF) option(MTGODOT_BUILD_PORT "Build only port_logic (with MTGODOT_BUILD_EXTENSION=OFF), e.g. for Linux/Windows" OFF) +option(MT_BUILD_NATIVE_RENDER_PROTOTYPE "Build the standalone Vulkan draw-command prototype" OFF) # --- shared, cross-platform core (no godot-cpp, no third-party) --- add_subdirectory(libgr2) @@ -81,3 +82,7 @@ if(MTGODOT_BUILD_PORT AND NOT MTGODOT_BUILD_EXTENSION) endif() add_subdirectory(extension/src/port) endif() + +if(MT_BUILD_NATIVE_RENDER_PROTOTYPE) + add_subdirectory(native_render) +endif() diff --git a/Client/mouse.cfg b/Client/mouse.cfg new file mode 100644 index 00000000..ba2682b7 --- /dev/null +++ b/Client/mouse.cfg @@ -0,0 +1 @@ +2 3 \ No newline at end of file diff --git a/audit/history.jsonl b/audit/history.jsonl index 0d891b4a..57ca3dfe 100644 --- a/audit/history.jsonl +++ b/audit/history.jsonl @@ -555,3 +555,5 @@ {"time": "2026-09-25T03:00:00Z", "event": "route_cleanup", "unit": "client_main legacy elimination", "detail": "Removed legacy AppFlow route, preloads (AppFlow, LiveSmokeTest, PlayableLiveTest, NetTrace), and MT_PORT_PATH fallback from project/client_main.gd. client_main.tscn now unconditionally boots the native 40250 port route (PythonUISurface + Python3DSurface -> system.py). Passing MT_PORT_PATH=legacy emits a deprecation warning and continues on port route.", "tests": ["client_main.tscn verified on port route and legacy warning", "script/port_gate.sh macos PASS (19/19 suites)", "port_map.py check: 0 errors"]} {"time": "2026-09-24T07:07:19Z", "event": "port_verification_fix", "unit": "Client/root/*.py + UserInterface/PythonApplication platform adapter + MilesLib/SoundManager platform adapter", "detail": "Corrected 74 RUN_AS_IS mappings from unrelated assets/root to 40250 root.epk, added per-script byte/hash verification and stricter port-map validation. Restored GC_TIME server clock semantics and routed packed sound bytes to Godot without applying volume twice; replaced deleted legacy asset-gate tests with active host/audio tests. Qualified MT_PLATFORM_STUB_TRACE exposed 93 called stubs in offline login_flow; SetGlobalCenterPosition was implemented from the original call order, and remaining reached groups were recorded in audit/reports/platform-runtime-trace.md. Android port-only gate exposed the pre-existing Win32Crt.cpp glob/iconv API-24 build blocker after a missing PCH dependency in Japanese.cpp was repaired.", "tests": ["script/verify_root_pack.py PASS (74/74)", "port_map.py check: 0 errors", "script/port_gate.sh macos PASS", "script/run_40250_asset_gate.sh PASS", "script/python_game_render_test.sh PASS", "MT_PLATFORM_STUB_TRACE=1 port.login_flow PASS (93 called stubs identified)", "android port_gate FAIL: Win32Crt.cpp glob/iconv declarations under NDK API 24"]} {"time": "2026-09-24T07:31:08Z", "event": "platform_adapter_fix", "unit": "EterLib/GrpScreen.cpp + MilesLib/SoundManager.cpp", "detail": "Implemented reached 40250 UI box and vertical gradient command bridge; audio bridge now preserves named three-slot music fades, frame-step fade speed and limit, indexed 3D volume, 3D play counts, and SaveVolume/RestoreVolume mute state. Remaining sky/water/shadow/culling/IME adapters stay TODO.", "evidence": ["./script/port_gate.sh macos", "./script/python_game_render_test.sh", "godot --headless --path project -s res://audio_driver_test.gd", "port_map.py check"]} +{"date": "2026-09-28T01:31:59+00:00", "unit": "GameLib/MapOutdoorRenderHTP.cpp + GameLib/MapOutdoorCharacterShadow.cpp", "action": "native 40250 render parity: terrain near/mid/far LOD/fog/splat cap, character shadow RTT CPU raster, per-draw depth/sampler states and two-light fixed-function shading", "evidence": "mt_native_render fake-server game 90 frames; 178 3D draws; shadow mem texture 205 nonwhite pixels; matched Windows screenshot pending"} +{"date": "2026-09-28T01:35:06+00:00", "unit": "GameLib/MapOutdoorRenderHTP.cpp", "action": "match 40250 lighting state in terrain splat passes: unlit base textures, temporary lit shadow pass, white fog color only for shadow pass", "evidence": "reference MapOutdoorRenderHTP.cpp non-WORLD_EDITOR branch; native fake-server scene rerun pending"} diff --git a/audit/manifest.json b/audit/manifest.json index c714cc98..48cf533f 100644 --- a/audit/manifest.json +++ b/audit/manifest.json @@ -4049,11 +4049,11 @@ "project/gamescene_test.gd", "project/netbridge_test.gd" ], - "last_test_result": "PARTIAL: msenv/skybox/weather tests, xmas_snow native command, GameScene map filtering and M2Client signal pass; SnowEnvironment pool, DayMode, filter/lens flare/wind and linear fog remain open" + "last_test_result": "PARTIAL: native fake-server 1024x768/24-mob render built and ran 90 frames; terrain and R5G6B5 character shadow target produced scene draws and 204 nonwhite shadow pixels. Same-scene Windows color parity, SnowEnvironment blur, DayMode, filter/lens flare/wind and fog timing remain open." }, "remaining": [ "实现或明确 screen filter、lens flare 和 SpeedTree/树木风更新,不以 metadata 代替运行时效果", - "修正/验证 40250 线性 fog near/far/color 与 HTP/STP/terrain 的参数和时序", + "原生 HTP 已恢复近/中/远景分段、LOD、splat 上限及远景雾色;继续用 Windows 同场景截图核对颜色、STP 和 fog 时序", "补齐 DayMode/PRESERVE_DayMode 正式环境切换;WeatherDayNightSystem 的自动地图预设继续隔离,不能替代服务器 xmas_snow 命令", "对天空 transition、云 plane、SnowEnvironment 池和真实渲染顺序增加参数/screenshot oracle", "增加重复加载、快速切图、暂停恢复、资源缺失和环境/粒子清理测试", diff --git a/audit/port-map/GameLib/MapOutdoorCharacterShadow.cpp.json b/audit/port-map/GameLib/MapOutdoorCharacterShadow.cpp.json index 1fb7671c..2587a9f7 100644 --- a/audit/port-map/GameLib/MapOutdoorCharacterShadow.cpp.json +++ b/audit/port-map/GameLib/MapOutdoorCharacterShadow.cpp.json @@ -9,7 +9,7 @@ "impl": [ "extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp:CMapOutdoor::SetShadowTextureSize" ], - "note": "40250 body verbatim, kept in the platform file beside the render-to-texture pass it serves (Begin/End are platform no-ops).", + "note": "40250 shadow texture lifecycle in the platform file; Begin/End now run the light-view offscreen pass through the RecordingDevice CPU R5G6B5 target.", "test": "extension/tests/port_login_flow_test.cpp" }, "CMapOutdoor::CreateCharacterShadowTexture": { @@ -17,7 +17,7 @@ "impl": [ "extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp:CMapOutdoor::CreateCharacterShadowTexture" ], - "note": "40250 body verbatim, kept in the platform file beside the render-to-texture pass it serves (Begin/End are platform no-ops).", + "note": "40250 shadow texture creation retained; the RecordingDevice now rasterizes its R5G6B5 render target and exposes the result as a memory texture.", "test": "extension/tests/port_login_flow_test.cpp" }, "CMapOutdoor::ReleaseCharacterShadowTexture": { @@ -25,22 +25,22 @@ "impl": [ "extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp:CMapOutdoor::ReleaseCharacterShadowTexture" ], - "note": "40250 body verbatim, kept in the platform file beside the render-to-texture pass it serves (Begin/End are platform no-ops).", + "note": "40250 shadow texture release retained for the native offscreen pass.", "test": "extension/tests/port_login_flow_test.cpp" }, "CMapOutdoor::BeginRenderCharacterShadowToTexture": { - "status": "DIVERGENT", + "status": "NEEDS_LIVE", "impl": [ "extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp:CMapOutdoor::BeginRenderCharacterShadowToTexture" ], - "note": "Returns false: the render-to-texture character shadow pass is not drawn (RecordingDevice drops off-screen render targets); callers skip the pass as when 40250 fails to begin it." + "note": "Restores 40250 light view, target, viewport and render states; the small R5G6B5 target is rasterized on CPU. Native fake-server capture contains nonwhite shadow pixels; matched Windows screenshot parity remains open." }, "CMapOutdoor::EndRenderCharacterShadowToTexture": { - "status": "DIVERGENT", + "status": "NEEDS_LIVE", "impl": [ "extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp:CMapOutdoor::EndRenderCharacterShadowToTexture" ], - "note": "No-op, paired with BeginRenderCharacterShadowToTexture returning false." + "note": "Restores the 40250 viewport, render target, light state and transforms after the shadow pass; matched Windows screenshot parity remains open." } } } diff --git a/audit/port-map/GameLib/MapOutdoorRenderHTP.cpp.json b/audit/port-map/GameLib/MapOutdoorRenderHTP.cpp.json index 15b560f3..9e0e5758 100644 --- a/audit/port-map/GameLib/MapOutdoorRenderHTP.cpp.json +++ b/audit/port-map/GameLib/MapOutdoorRenderHTP.cpp.json @@ -9,17 +9,21 @@ "impl": [ "extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp:CMapOutdoor::__RenderTerrain_RenderHardwareTransformPatch" ], - "note": "No-op: the Godot terrain adapter draws the height-field (PORT-PLAN §3); the D3D8 texture-stage splat passes have no recording-device counterpart." + "note": "Native path now follows 40250 near/mid/far patch distance bands, LOD 0/1/2, splat limit and fog-color far fill; Godot path stays adapted. Same-scene Windows color and image comparison remains open." }, "CMapOutdoor::__HardwareTransformPatch_RenderPatchSplat": { "status": "DIVERGENT", - "impl": [], - "note": "Not built: only called from __RenderTerrain_RenderHardwareTransformPatch, which is a no-op (Godot draws the terrain, PORT-PLAN §3)." + "impl": [ + "extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp:CMapOutdoor::__HardwareTransformPatch_RenderPatchSplat" + ], + "note": "Native path draws 40250 terrain splat layers and static/character shadows; texture-0 fallback and state sequencing still differ from the reference and need matched image verification." }, "CMapOutdoor::__HardwareTransformPatch_RenderPatchNone": { - "status": "DIVERGENT", - "impl": [], - "note": "Not built: only called from __RenderTerrain_RenderHardwareTransformPatch, which is a no-op (Godot draws the terrain, PORT-PLAN §3)." + "status": "NEEDS_LIVE", + "impl": [ + "extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp:CMapOutdoor::__HardwareTransformPatch_RenderPatchNone" + ], + "note": "Native path now draws the far patch with the 40250 fog texture factor; matched Windows image comparison remains open." } } } diff --git a/docs/CLIENT-PERFORMANCE-INVESTIGATION.md b/docs/CLIENT-PERFORMANCE-INVESTIGATION.md new file mode 100644 index 00000000..1c668cff --- /dev/null +++ b/docs/CLIENT-PERFORMANCE-INVESTIGATION.md @@ -0,0 +1,298 @@ +# 客户端运行流畅度与性能瓶颈调查报告 + +> **调查日期**:2026-09-24 +> **调查目标**:定位 macOS 客户端在运行、移动与战斗过程中出现不流畅、微卡顿(micro-stutter)与跳帧感的根本原因,提出针对性优化方案。 +> **状态**:已完成调查,尚未引入代码改动。 + +--- + +## 一、 核心结论概述 + +经链路排查,客户端运行不流畅的**根本原因不在于 40250 原版游戏逻辑本身的性能**,而在于当前 **Godot 与 40250 C++ 桥接管线中存在的 6 大性能与同步瓶颈**: + +1. **显存颠簸(致命瓶颈)**:每一帧都在销毁并重新创建数十到上百个 GPU 网格缓冲(`ArrayMesh.new()`),macOS Metal 驱动层发生每秒数千次的显存分配与释放(Memory Allocator Thrashing)。 +2. **跨语言数据巨量重复封包与无意义消耗**:每帧将数万顶点、法线、矩阵打包为 Godot Dictionary,且在 UI 表面仅仅为了判断 `has_3d` 就把全量顶点数据封包一次并立即丢弃。 +3. **Mac 120Hz ProMotion 与 40250 60 FPS 逻辑时钟失步**:未设置帧率上限导致以 120Hz 刷新,每帧时间在 8ms 与 16ms 之间跳跃,不仅引发明显的视觉顿挫感,还导致所有 CPU/GPU 颠簸开销翻倍。 +4. **GDScript 解释执行高频 CPU 计算**:脚本层三重循环执行 4x4 矩阵乘法、逐 UI 图元无条件执行复杂多边形布尔裁剪。 +5. **双线程每帧 Ping-Pong 等待**:Godot 主线程与 Python Script 线程通过互斥锁与条件变量每帧往返切换,受 macOS 大小核调度抖动影响。 +6. **Retina 屏上的重型渲染器配置**:默认启用了 `Forward+` 与 `MSAA 2x`,在高 DPI 屏幕上带来了不必要的渲染计算与显存带宽压力。 + +--- + +## 二、 关键瓶颈深度剖析 + +### 1. 致命瓶颈:每帧反复创建并销毁 `ArrayMesh`(显存颠簸) +* **代码位置**:`project/python_3d_surface.gd:121, 342-356` +* **实现逻辑**: + ```gdscript + func update_frame() -> void: + ... + instance.mesh = _build_mesh(draw) # 每帧对每个 3D 绘制对象调用 + + func _build_mesh(draw: Dictionary) -> ArrayMesh: + var arrays := [] + arrays.resize(Mesh.ARRAY_MAX) + arrays[Mesh.ARRAY_VERTEX] = draw["positions"] + ... + var mesh := ArrayMesh.new() + mesh.add_surface_from_arrays(Mesh.PRIMITIVE_TRIANGLES, arrays) + mesh.surface_set_material(0, _material(draw)) + return mesh + ``` +* **机制危害**: + - 在 Godot 架构中,`ArrayMesh.new()` 配合 `add_surface_from_arrays` 会在底层图形 API(macOS Metal / Vulkan)中**新申请显存分配顶点缓冲区(VBO)和索引缓冲区(IBO)**。 + - 场景中包含角色各部位(身体、武器、头发)、怪物、NPC、特效等数十至上百个 draw call。下一帧到来时,上一帧的所有 `ArrayMesh` 被丢弃进 GC 释放显存。 + - **以 60 FPS 计,每秒发生 3,000 ~ 6,000 次 GPU 显存分配和释放**;若在 120Hz 屏幕上更达 **6,000 ~ 12,000 次/秒**。 + - 这会引发 macOS Metal 驱动内存分配器严重颠簸(Memory Allocator Thrashing)和驱动锁争用,是掉帧与微卡顿的最大源头。 + +--- + +### 2. C++ 到 GDScript 巨量跨语言数据封包与重复调用 +* **代码位置**: + 1. `project/python_ui_surface.gd:192-194` + 2. `project/python_3d_surface.gd:89, 136` + 3. `extension/src/python_host_node.cpp:169-220` +* **机制危害**: + - `Metin2PythonHost::render3d_draws()` 每次调用都会把所有 3D draw call 的 16 维矩阵、几万个顶点、法线、UV、颜色、索引逐个转换成 `PackedVector3Array`、`PackedFloat32Array` 并封装成带有 20~30 个键值对的庞大 Godot `Dictionary` 数组。 + - **严重重复调用**: + 在 `python_ui_surface.gd:192-194` 中: + ```gdscript + var draws_3d: Array = Metin2PythonHost.render3d_draws() + var has_3d: bool = not draws_3d.is_empty() + ``` + 这里**仅仅为了检查 3D 绘制队列是否为空**,就把整个场景的几何数据完整跨语言序列化了一遍,然后**立即全部丢弃进垃圾回收**! + 紧接着在 `python_3d_surface.gd:89` 又完整调用了一遍 `render3d_draws()` 真正用于绘制。 + - 此外,`Metin2PythonHost.ui_render_commands()` 每一帧也在 `python_ui_surface.gd:195` 与 `python_3d_surface.gd:136`(`_update_background()`)中**被调用了两次**。 + +--- + +### 3. 屏幕刷新率(ProMotion 120Hz)与 40250 逻辑时钟(60 FPS)失步 +* **代码位置**: + - `project/project.godot`(未配置 `max_fps`) + - `extension/src/port/EterBase/Timer.cpp:109-116` +* **机制危害**: + - MacBook 屏幕通常为 120Hz ProMotion。当前 `project.godot` 没有设置帧率上限,Godot 默认以 120 FPS 运行 `_process`。 + - 这导致上述所有 C++ 内存封包、显存频繁重建、线程唤醒**以每秒 120 次的频率翻倍运行**。 + - 更关键的是,40250 原版内部逻辑是以 60 FPS(约 16.6ms)的时钟基准设计的。在 120Hz 下,传递给 40250 的单帧 delta 时间约为 8.3ms,导致游戏内部插值在 8ms 与 16ms 之间来回跳动,产生肉眼可见的“画面抽动 / 跳帧”感。 + +--- + +### 4. GDScript 解释器高频数学与几何计算 +* **代码位置**: + 1. `project/python_3d_surface.gd:77-86`: + ```gdscript + static func multiply(a: PackedFloat32Array, b: PackedFloat32Array) -> PackedFloat32Array: + var out := PackedFloat32Array() + out.resize(16) + for r in 4: + for c in 4: + var sum := 0.0 + for k in 4: + sum += a[r * 4 + k] * b[k * 4 + c] + out[r * 4 + c] = sum + return out + ``` + 每一帧在 GDScript 解释器层面用三重循环对每一个 3D 模型进行 4x4 矩阵乘法,并频繁分配 16 个元素的数组。 + 2. `project/python_ui_surface.gd:341-355`: + `_clip_quad` 对界面上每个图元都无条件调用 `Geometry2D.intersect_polygons(...)` 进行多边形相交求值,即使 95% 以上的 UI 元素完全位于屏幕范围内未发生任何裁剪。 + +--- + +### 5. 双线程每帧条件变量往返切换(Ping-Pong Jitter) +* **代码位置**:`extension/src/platform/ScriptLib/PythonBoot.cpp:597, 612-616` +* **机制危害**: + Godot 主线程与 Python Script 线程通过 `g_fiber.cv.notify_all()` 与 `cv.wait()` 互斥锁交替执行。主线程渲染完必须挂起等待 Script 线程完成 `Process()`,若 macOS 系统调度器将其中一个线程调度到了能效核(E-Core)或与其他系统后台任务竞争,会造成帧间隔不均匀(Frame Time Jitter)。 + +--- + +### 6. 渲染器配置过重(Forward+ & MSAA 2x) +* **代码位置**:`project/project.godot:40-41` + ```ini + renderer/rendering_method="forward_plus" + anti_aliasing/quality/msaa_3d=2 + ``` +* **机制危害**: + Metin2 模型是典型的 2004 年代 Direct3D 8 单方向光、无复杂 PBR 的固定管线模型。在 Mac Retina 高分辨率(例如 3K/4K 物理分辨率)下,Godot 的 `Forward+` 会执行完整的光照聚类裁剪(Clustered Shading Compute Shaders),且 `MSAA 2x` 在高分辨率下会产生巨大的 Resolve 显存带宽负担。 + +--- + +## 三、 优化实施方案与优先级建议 + +| 阶段 | 优化措施 | 涉及文件 | 预期成效 | +| :--- | :--- | :--- | :--- | +| **P0** | **锁定 60 FPS 帧率**
在 `project.godot` 配置 `application/run/max_fps=60` | `project/project.godot` | 彻底消除 120Hz/60FPS 帧率抖动与跳帧感,立竿见影减少一半 CPU/GPU 压力与发热 | +| **P0** | **消除跨语言重复调用**
1. C++ 暴露轻量 `Metin2PythonHost.has_3d_draws()`,UI 表面不再调用 `render3d_draws()` 判空。
2. 缓存或单次提取 `ui_render_commands()`,避免同帧重复调用。 | `extension/src/python_host_node.cpp`
`project/python_ui_surface.gd`
`project/python_3d_surface.gd` | 消除每帧数万个无用 Variant 的内存分配与 GC 压力 | +| **P1** | **网格缓存与显存复用**
对动态 mesh 进行池化或直接利用动态更新,避免每帧 `ArrayMesh.new()` 导致 GPU VBO/IBO 频繁申请与释放。 | `project/python_3d_surface.gd` | **彻底根治 Metal 显存分配器颠簸**,使画面运行保持平滑稳定 | +| **P1** | **UI 裁剪短路优化**
若图元 quad 的外接矩形完全在屏幕与 clip 矩形之内,直接跳过 `Geometry2D.intersect_polygons` 计算。 | `project/python_ui_surface.gd` | 显著减轻 GDScript 解释器的 CPU 计算负担 | +| **P2** | **渲染器与抗锯齿轻量化**
评估切换为 `mobile` 渲染模式,在高 PPI Retina 屏上关闭 MSAA 2x 或改用开销更低的抗锯齿方案。 | `project/project.godot` | 大幅度降低 GPU 填充率与显存带宽消耗 | + +--- + +## 四、 已实施优化与复测对比(2026-09-25) + +### 优化轮次:`ui_draw` 专项优化 +1. **C++ 端预分配与 StringName 优化**:在 `extension/src/python_host_node.cpp` 中预分配 Command Array 空间,使用静态 `StringName` 替换所有字典字符串键,避免每帧数百次重复字符串分配与动态哈希运算。 +2. **纹理元数据与 Atlas 缓存**:在 `project/python_ui_surface.gd` 建立 `_tex_info_cache`,缓存纹理 RID、Atlas UV 缩放/偏移与全图默认 UV 数组,将 Atlas 区域与大小计算彻底移出每帧绘制主循环。 +3. **图元包围盒快速裁剪**:基于预计算的 `x1, y1, x2, y2` 与裁剪视口 `clip_x1, clip_y1, clip_x2, clip_y2` 执行 4 次浮点比较: + - 完全在视口外的图元直接跳过; + - 完全在视口内的图元(占比 >95%)走无裁切快路径,直接组装多边形顶点,完全跳过多边形求交计算与中间数组构造。 +4. **颜色数组与画布清理池化**:缓存高频单色数组,避免 `PackedColorArray([color])` 的动态堆分配;仅清除当帧实际使用过的 CanvasItem 节点。 + +### 复测性能对比(稳态场景 240 帧采样) + +| 性能指标 | 优化前 | 优化后 | 改善幅度 | +| :--- | :--- | :--- | :--- | +| **`ui_draw` 耗时 (P50)** | **9.267 ms** | **1.798 ms** | **-80.6%(减少 7.47 ms)** | +| **单帧平均耗时 (Mean)** | 21.126 ms (47.3 FPS) | **16.666 ms (60.0 FPS)** | **-21.1%(稳定跑满 60 帧)** | +| **单帧中位数耗时 (P50)** | 20.964 ms | **16.667 ms** | **-20.5%** | +| **95分位耗时 (P95)** | 22.391 ms | **16.920 ms** | **-24.4%(抖动几乎消除)** | +| **测试门禁状态** | PASS | **PASS (12/12 全绿)** | 画面与逻辑 100% 保持一致 | + +### 多怪战斗复测与字形批处理(2026-09-25) + +用户已在 40250 原版客户端实测相同怪物量运行流畅。因此,多怪场景的卡顿应优先排查移植层增加的开销,不能仅归因于原版逐字绘制或怪物数量。 + +`MT_FAKE_MOB_COUNT=24` 场景中,原始 UI 命令约 1000 条,其中约 900 条是字体图集 `mem:1@51` 的字形。Godot 端逐条将这些字形转换为画布多边形,是移植后额外的每帧开销。现在 C++ 在不改变命令顺序的前提下,把同纹理、无遮挡裁剪、无蒙版的相邻字形合并为一次三角形数组提交;其他 UI 图元沿用原路径。3D 背景提取也复用同帧批处理命令。 + +| 24 怪场景指标 | 批处理前 | 批处理后 | +| :--- | ---: | ---: | +| 空闲帧 P50 | 27.424 ms | 17.790 ms | +| 空闲帧 P95 | 28.788 ms | 19.221 ms | +| 空闲帧 UI 绘制 P50 | 7.253 ms | 1.086 ms | +| 战斗帧 P50 | 22.429 ms | 16.632 ms | +| 战斗帧 UI 绘制 P50 | 6.470 ms | 1.063 ms | + +批处理后的一帧诊断中,约 1000 条原始 UI 命令缩减为 345 条提交命令,其中 7 个批次容纳 666 个字形。单怪场景的空闲与战斗帧 P50 均约 16.66 ms。24 怪场景的登录、进图、贴地、贴图、光照、截图及击杀检查通过。空闲帧 P50 仍约 17.8 ms,`ui_update` P50 约 11.6 ms,是后续分析重点。以上数字来自本地自动化场景,不能当作原版与移植版在同一机器上的直接性能对比。 + +原始和复测数据分别在 `build/rendering/python-game-20260925-064436-59271/report.json` 与 `build/rendering/python-game-20260925-072043-60686/report.json`。 + +### 64 怪压力测试(2026-09-25) + +`MT_FAKE_MOB_COUNT=64` 的本地测试通过登录、进图、画面及击杀检查;截图检查时视野内记录到 38 个角色。无额外原生计时器时,空闲 240 帧的帧间隔 P50 为 **27.028 ms**、P95 为 **28.785 ms**(约 37 FPS);战斗采样 P50 为 **16.732 ms**、P95 为 **21.449 ms**。战斗段的可见内容和角色数量会变化,因此不能凭其 P50 宣称 64 怪稳定 60 FPS。数据见 `build/rendering/python-game-20260925-072444-60992/report.json`。 + +空闲帧里,Godot 记录的 `ui_update` P50 为 17.146 ms、3D 表面更新 P50 为 6.292 ms、UI 画布绘制 P50 为 1.737 ms。`ui_update` 这个字段包含脚本线程执行完整的原生 `CPythonApplication::Process()`,并不等于纯界面更新。临时细分计时表明原生窗口树 `OnUIUpdate()` 只有约 0.3 ms,原生渲染录制约 16 ms;其中 `RenderGame()` 约 11 ms,角色 `Deform()` 约 5 ms,角色 `Render()` 约 5 ms,背景约 0.8 ms。这些是连续 120 帧的平均值,计时探针本身会增加少量开销;探针已移除,日志保存在 `build/rendering/python-game-20260925-072854-61643/godot.log`。 + +这一场景的原始 UI 图像命令为 1747 条,其中字体图集字形 1619 条;批处理后的命令总量 538 条,13 个字形批次覆盖 1229 个字形。后续优先分析角色变形和原生绘制录制的逐怪线性开销,再分析 Godot 3D 几何提交;继续优化 Godot UI 画布预计收益较小。任何角色动画降频、剔除或 LOD 都要先与 40250 原版可见行为对照。 + +### 渲染转换第一轮(2026-09-25) + +先试过对动画 `ArrayMesh` 原地更新;单怪测试中 3D 更新 P50 升至约 12 ms。临时细分计时发现每帧顶点打包和拓扑哈希使跨语言命令提取耗时约 14 ms,而且网格原地更新没有命中。该试验已撤回,没有保留在运行路径。 + +随后改为在 Godot 单精度构建中,把原生连续 `float` 顶点、法线和 UV 批量复制到对应 Packed 数组;非单精度构建保留逐元素转换。64 怪场景在相同临时计时下,命令提取 P50 从 **2.583 ms** 降至 **1.571 ms**,3D 表面更新 P50 从 **6.857 ms** 降至 **5.280 ms**,空闲帧 P50 从 **28.109 ms** 降至 **25.603 ms**;战斗 P50 仍约 16.7 ms。数据见 `build/rendering/python-game-20260925-075500-62940/report.json` 和 `build/rendering/python-game-20260925-075707-63151/report.json`。单怪测试空闲帧 P50 为 16.667 ms,24 怪两次复测 P50 为 18.639/18.547 ms;三档测试的画面和击杀检查均通过。24 怪整体帧时间比更早的 17.790 ms 测量略高,而本次测得的 3D 表面更新时间更低,差异主要出现在原生帧阶段,仍需重复采样判断。 + +移除临时细分计时器后,最终 64 怪复测的空闲帧 P50/P95 为 **25.591/27.372 ms**,3D 表面更新 P50 为 **5.240 ms**,战斗 P50/P95 为 **16.718/21.065 ms**,画面和击杀检查通过;见 `build/rendering/python-game-20260925-080029-63326/report.json`。这仍未达到 64 怪稳定 60 FPS。下一步若改用 C++ RenderingServer 或 GPU 蒙皮,应分别验证稳定网格身份与拓扑复用、骨骼权重和绑定姿态的导出,以及逐帧顶点/GPU 同步成本;不能把底层 API 本身当作零拷贝或固定收益保证。 + +### 批量转换与几何缓存完整验证(2026-09-25 最新复测) + +对当前未提交代码(含 C++ 连续内存批量 `memcpy` 复制、静态网格 `geometry_key` 缓存复用、UI 批处理、以及输入与拾取射线同步)进行了完整的 1 怪、24 怪、64 怪多密度验证: + +| 测试场景 | 空闲帧 P50 | 空闲帧 P95 | 3D 表面更新 P50 | UI 绘制 P50 | 战斗帧 P50 | 战斗 3D 更新 P50 | 门禁与击杀检查 | +| :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | +| **1 怪 (单体基准)** | **16.679 ms** (60 FPS) | 16.825 ms | **1.928 ms** | 0.653 ms | 16.670 ms | 1.664 ms | PASS (已击杀) | +| **24 怪 (中等密度)** | **17.611 ms** | 18.078 ms | **3.346 ms** | 1.061 ms | 16.655 ms | 2.499 ms | PASS (已击杀) | +| **64 怪 (极限压力)** | **25.619 ms** (约 39 FPS) | 26.493 ms | **5.252 ms** | 1.664 ms | 16.710 ms | 2.926 ms | PASS (已击杀) | + +**结论与评估**: +1. **优化切实有效且无低密度回退**: + - 1 怪场景的 3D 表面更新由最初的 2.3 ms 降至 1.93 ms,空闲与战斗均稳定在 60 FPS(16.67 ms),证明没有像第一版那样出现小场景负优化(之前第一版原地更新曾劣化到 11.9 ms)。 + - 24 怪场景空闲帧 P50 取得 17.611 ms、P95 取得 18.078 ms,创下历史最佳稳定性;3D 表面更新仅 3.35 ms,战斗全程维持 60 FPS(16.66 ms)。 + - 64 怪极端场景下,3D 表面更新由优化前基准的 6.857 ms 降至 5.252 ms(**减少 1.605 ms / -23.4%**);空闲帧由 28.109 ms 降至 25.619 ms(**减少 2.49 ms / -8.9%**)。 +2. **正确性完全保证**: + - 1 怪、24 怪、64 怪所有用例的地面贴地、材质贴图、光照、屏幕对照度分析、目标锁定与鼠标左键普攻击杀(HP 递减为 [100, 66, 33, 0])均 100% 通过。 + - 报告数据分别保存在: + - 1 怪:`build/rendering/python-game-20260925-090440-78310/report.json` + - 24 怪:`build/rendering/python-game-20260925-090521-78357/report.json` + - 64 怪:`build/rendering/python-game-20260925-090617-78378/report.json` + +### 独立 Vulkan/MoltenVK 原型(2026-09-25) + +新增 `native_render/`:SDL3 创建独立窗口,Vulkan 创建交换链、顶点缓冲、管线与逐 draw 提交;macOS 通过 MoltenVK 映射到 Metal。它直接读取录制设备的 `Render3DDraw` 结构,也能回放 Godot 客户端显式捕获的一帧。构建和回放命令见 `native_render/README.md`。现有 Godot 客户端运行路径未切换。 + +#### 第一阶段:基础提交与每帧 CPU 展开基线 +Apple M4 上,合成的 64 draw × 333 三角形(63,936 个展开顶点)运行 60 帧,平均帧间隔 16.64 ms,CPU 几何准备 6.80 ms。真实 64 怪测试捕获的一帧(`64-mob.mtdr`,v1)为 51 draw、51,231 个索引;回放 60 帧,平均帧间隔 16.81 ms,CPU 几何准备 5.96 ms。 + +#### 第二阶段:按 `geometry_key` 持久化 GPU 顶点/索引缓冲 + 着色器矩阵变换 +1. **持久化 GPU 缓冲与按修订号增量上传**:以 `GeometryId{geometry_key, signature}` 作为 GPU 缓冲缓存键(而非 draw 数组下标,避免视锥裁剪导致 draw 顺序变化时整表失效),仅在新几何出现或 `geometry_revision` 变化时上传顶点与索引缓冲。 +2. **着色器 `world * view * proj` 变换**:将每帧 CPU 顶点变换移入 `native.vert` Push Constants,索引缓冲直接走 `vkCmdDrawIndexed`(消除每帧 CPU 三角形展开)。 + +#### 第三阶段:深度缓冲、DDS 贴图、法线/UV 与 D3D8 固定管线状态(Version 3 捕获) +1. **深度与固定管线状态**:新增 `VK_FORMAT_D32_SFLOAT` 深度附件,按 `(cull_mode, z_enable/z_write, alpha_blend/dest_blend)` 缓存 `VkPipeline`,在 `native.vert` / `native.frag`(128 字节 Push Constants,满足桌面与 Android Vulkan 最小保证)中实现 D3D8 方向光(`light0` + 环境光 + 自发光)、`texture_factor` 调制与 `alpha_test` 裁剪。 +2. **自包含 DDS 贴图回放**:捕获格式升级为 Version 3(向下兼容 v1/v2),在 `MT_NATIVE_CAPTURE_PATH` 触发时自动从 40250 pack 提取被引用的 `.dds` 原始字节写入 `.mtdr`;`mt_native_render` 复用 `extension/src/dxt.cpp` 在首帧软解 DXT1/3/5 并上传至 `VK_IMAGE_TILING_OPTIMAL` 纹理与描述符集。 +3. **修复无 `MT_PROFILE_FRAME` 时 64 怪测试 `dog_screen=(-1,-1)` 失败的根因**: + - 根因一:`CPythonNetworkStream::GamePhase()` 每帧最多消费 3 个网络包,64 怪场景的进图突发包(约 147 个包,含末尾 `GC_PING`)需要约 49 帧排空;此前未开启 `MT_PROFILE_FRAME` 时仅等待 30 帧就点击地面移动,导致客户端在回复 `CG_PONG`(254)前先发出了 `CG_CHARACTER_MOVE`(7),伪造服务端报 `expected CG header 254, got 7` 并主动断开连接清空角色。 + - 根因二:`settled` 闭包未检查 `now.x >= 0.0`,连续两帧 `(-1, -1)` 距离为 `0.0 < 0.5` 会提前退出等待。 + - 修复后,`MT_FAKE_MOB_COUNT=64` 在不开启 `MT_PROFILE_FRAME` 的捕获运行中也 100% 通过所有贴图、光照、投影与普攻击杀检查(`build/native-render/capture-64mob-v3/report.json`)。 + +#### Apple M4 原生 Vulkan 60 帧回放实测对比 + +| 测试用例(60 帧,FIFO 垂直同步) | Draw 数 | 顶点数 | 索引数 | 60 帧总几何上传 | 贴图上传 | 全帧平均 `prepare_ms` | **稳态 `steady_prepare_ms`(第 2~60 帧)** | `submit_ms` | `mean_frame_ms` | +| :--- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| **合成 64 draw(阶段一:每帧 CPU 展开)** | 64 | 63,936 | — | 3,840 次 | 0 | 6.80 ms | 6.80 ms | 8.21 ms | 16.64 ms | +| **真实 64 怪 v1(阶段一:每帧 CPU 展开)** | 51 | 51,231 | — | 3,060 次 | 0 | 5.96 ms | 5.96 ms | 9.33 ms | 16.81 ms | +| **合成 64 draw(阶段二/三:持久 GPU 缓冲)** | 64 | 63,936 | 63,936 | **64 次** (3.32 MB) | 0 | 0.17 ms | **0.13 ms** | 13.88 ms | 16.60 ms | +| **真实 64 怪 v2(51 draw,持久 GPU 缓冲)** | 51 | 216,384 | 51,231 | **51 次** (10.59 MB) | 0 | 0.21 ms | **0.11 ms** | 14.13 ms | 16.35 ms | +| **真实 64 怪 v3(90 draw,含 17 张 DDS 贴图 + 光照 + 深度)** | 90 | 244,520 | 157,647 | **90 次** (12.37 MB) | **17 次** (18.37 MB) | 2.48 ms(含首帧冷启动解纹理) | **0.23 ms** | 13.57 ms | 18.75 ms | +| **真实 64 怪 v3 + 首 draw 每帧动画修订(`--animate-first-draw`)** | 90 | 244,520 | 157,647 | **149 次** (90 + 59) | **17 次** (18.37 MB) | 2.66 ms(含首帧冷启动解纹理) | **0.43 ms** | 13.45 ms | 18.82 ms | + +#### 第四阶段:硬件 GPU 时间戳、40250 地形 HTP、2D UI/字形/小地图遮罩、GPU 骨骼蒙皮与实时游戏主循环接入(Version 4 捕获) + +1. **分离真实 GPU 执行耗时(`gpu_ms`)与垂直同步等待(`submit_ms`)**: + - 在 `native_render/main.cpp` 中接入 `VkQueryPool`(`VK_QUERY_TYPE_TIMESTAMP`),在 `vkCmdBeginRenderPass`(`TOP_OF_PIPE`)与 `vkCmdEndRenderPass`(`BOTTOM_OF_PIPE`)记录硬件时间戳并换算为毫秒级 `gpu_ms`。 + - 新增 `--no-vsync`(优先选用 `VK_PRESENT_MODE_IMMEDIATE_KHR` / `MAILBOX_KHR`),将交换链垂直同步等待从 `submit_ms` 中剥离。 +2. **40250 原生硬件变换地形块(HTP)渲染**: + - 在 `extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp` 中实现 `CMapOutdoor::__RenderTerrain_RenderHardwareTransformPatch()` 与 `__HardwareTransformPatch_RenderPatchSplat(...)`(由 `SetNativeTerrainRenderEnabled(true)` 或 `MT_NATIVE_TERRAIN=1` 启用,不影响 Godot 默认测试路径)。 + - 在 `extension/src/platform/EterLib/RecordingDevice.cpp` 中支持 D3D8 `D3DTSS_TCI_CAMERASPACEPOSITION` 相机空间纹理矩阵生成地形基础层与 Alpha Splat 层 UV,并将 `m_textures[0]` 混入 `geometry_key` 实现跨帧持久化缓存。 +3. **2D `UIRenderCommand` 全量渲染(含字形图集与小地图圆角遮罩)**: + - 支持 `behind_3d` 与前景两趟正交 2D 渲染,涵盖 `Bar`、`GradientBar`、`Line`、`Image`、裁剪矩形(`clip_x1..y2`)、`mem:@` 动态字形图集(`"MTRA"` 原始 RGBA 编码)、`.tga`(含未压缩与 RLE)解码,以及 `CPythonMiniMap` 的双纹理(`tex0` + `mask_tex`)圆角遮罩混合。 + - 相邻同状态、同纹理的 UI 图元与字形自动合并为单次 `vkCmdDrawIndexed` 批次提交(64 怪场景 1,597 个 UI/字形四边形成批为 31 个 Vulkan 批次)。 +4. **GPU 线性混合骨骼蒙皮(GPU Skeletal Skinning)**: + - 在 `extension/src/platform/EterGrnLib/GrannyRuntime.cpp` 中,当启用 GPU 蒙皮(`--gpu-skinning` 或 `MT_GPU_SKINNING=1`)时,`GrannyDeformVertices` 不再逐帧在 CPU 上对每个顶点做矩阵加权变形,而是按不可变的源网格指针(`SourceVertices`)提取一次静态绑定姿态(Bind-Pose)顶点、法线、4 骨骼索引(`bone_indices`)与 4 骨骼权重(`bone_weights`),仅将当前实例的骨骼矩阵调色板(`bone_count * 16` 个 `float`)写入切片表。 + - 在 `RecordingDevice.cpp` 中,所有共享同一 `.gr2` 源网格的怪物/角色实例共享同一个 `geometry_key` 且 `geometry_revision` 恒为 `1`;`native.vert` 通过 `set = 1, binding = 0` 的 `BonePalette` SSBO 在顶点着色器中完成 4 骨骼线性混合蒙皮。 +5. **实时游戏主循环直接接入(`--live-client`)**: + - 当 `mt_native_render` 在主构建目录 `build/` 下构建时,直接链接 `port_platform`、`mtpython` 与 `FakeLoginServer`,通过 `--live-client ` 在原生 SDL3 + Vulkan 窗口内直接启动 40250 `system.py`,完成登录、选人、进图、刷出 1~64 只怪物、转发 SDL3 键鼠事件,并可导出包含 3D 骨骼权重、地形、2D UI 与字形页的 Version 4 `.mtdr` 捕获文件。 + +#### Apple M4 第四阶段实测数据(实时客户端 `--live-client` 与 v4 回放,60 帧) + +| 运行模式(Apple M4,60 帧) | 呈现模式 | 3D Draw (含地形) | GPU 蒙皮 Draw | UI 批次 / 四边形 | 顶点数 / 索引数 | **60 帧动态几何重传** | **`game_update_ms` (40250 逻辑)** | **稳态 `steady_prepare_ms`** | **`submit_ms`** | **硬件 `gpu_ms`** | **`mean_frame_ms`** | +| :--- | :--- | ---: | ---: | ---: | ---: | :--- | ---: | ---: | ---: | ---: | ---: | +| **实时 1 怪 (`--live-client --gpu-skinning --no-vsync`)** | `IMMEDIATE` | 74 | 10 | 31 / 220 | 225,355 / 91,119 | 30 次 (9.8 KB) | 5.55 ms | **0.62 ms** | 2.22 ms | **0.36 ms** | **8.48 ms** (~118 FPS) | +| **实时 64 怪・CPU 蒙皮 (`--live-client --no-gpu-skinning --no-vsync`)** | `IMMEDIATE` | 116 | 0 | 31 / 1,597 | 253,999 / 207,543 | **3,178 次 (205.37 MB)** | 19.93 ms | 3.42 ms | 0.16 ms | **0.83 ms** | 23.62 ms (~42 FPS) | +| **实时 64 怪・GPU 蒙皮 (`--live-client --gpu-skinning --no-vsync`)** | `IMMEDIATE` | 117 | **52** | 31 / 1,597 | 254,003 / 207,549 | **28 次 (9.18 KB)** | **10.51 ms** | **0.78 ms** | **0.10 ms** | **0.79 ms** | **11.46 ms (~87 FPS)** | +| **实时 64 怪・GPU 蒙皮 (`--live-client --gpu-skinning`,默认 FIFO)** | `FIFO` | 116 | **52** | 31 / 1,597 | 253,999 / 207,543 | **19 次 (6.23 KB)** | **10.42 ms** | **0.79 ms** | **0.10 ms** | **0.88 ms** | **11.37 ms** | +| **v4 捕获回放 + 骨骼动画 (`--capture 64-mob-v4.mtdr --animate-bones --no-vsync`)** | `IMMEDIATE` | 116 | **52** | 31 / 1,597 | 253,999 / 207,543 | **首帧 74 次,第 2~60 帧 0 次** | — | **1.11 ms** | 5.19 ms | **1.08 ms** | 12.25 ms | + +**关键验证结论**: +1. **真实 GPU 耗时(`gpu_ms`)极低**:通过 Vulkan 硬件时间戳确认,Apple M4 执行完整 64 怪场景(117 个 3D Draw 含地形多级 Splat、52 个 GPU 骨骼蒙皮网格、31 个 UI 批次共 1,597 个 UI/字形四边形、约 25.4 万顶点)的单帧真实 GPU 耗时仅为 **0.79 ~ 0.88 ms**,此前 `submit_ms` 的 ~13.5 ms 完全来自 `FIFO` 垂直同步等待。 +2. **GPU 骨骼蒙皮消除 99.995% 动态顶点带宽并减半 CPU 帧耗时**:在 64 怪实时客户端中,开启 `--gpu-skinning` 后: + - 60 帧动态几何上传量从 CPU 蒙皮的 **3,178 次 / 205.37 MB** 降至 **28 次 / 9.18 KB**(所有同模型怪物实例共享同一份静态绑定姿态 GPU 缓冲); + - 40250 游戏帧更新 `game_update_ms` 从 **19.93 ms** 降至 **10.51 ms**(每帧节省 **9.42 ms**); + - 渲染准备 `steady_prepare_ms` 从 **3.42 ms** 降至 **0.78 ms**(每帧节省 **2.64 ms**); + - 端到端单帧总耗时 `mean_frame_ms` 从 **23.62 ms(~42 FPS)** 降至 **11.46 ms(~87 FPS)**,在 Debug 构建下即已稳定跑进 60 FPS(16.67 ms)预算以内。 + +#### 第五阶段:完整可玩原生客户端补齐(水面/天空盒/云层/SpeedTree/球面环境高光、音频、全键鼠/IME/软件光标、交互启动与 Release `-O3` 构建) + +按路线图顺序完成全部 5 项剩余原生特性(通过 `IsNativeTerrainRenderEnabled()` 门控新 3D 要素,保持 Godot 默认路径 32/32 回归检查 100% 通过): +1. **交互启动与窗口/贴图扩展**: + - 新增 `--interactive`(无限帧循环直至关闭窗口)、`--login-screen`(停留在 `introLogin.LoginWindow` 供手动输入账号密码)、`--live-server HOST:AUTH_PORT:GAME_PORT`(直连外部 40250 服务端)与 `--width W --height H`。 + - 支持 `SDL_WINDOW_RESIZABLE` 与 `VK_ERROR_OUT_OF_DATE_KHR` / `VK_SUBOPTIMAL_KHR` 交换链自动重建,并同步调用 `PythonBoot::SetUISize`。 + - 接入 macOS `ImageIO` 解码器,支持从 40250 pack 直接解码 `.jpg` / `.png` / `.bmp` 登录背景与 UI 贴图;并在未显式设置 `VK_ICD_FILENAMES` 时自动探测 Homebrew `MoltenVK_icd.json`。 +2. **补齐剩余 3D 场景视觉要素**: + - **水面(Water)**:移植 `extension/src/platform/GameLib/MapOutdoorWater.cpp`(`CMapOutdoor::RenderWater` 与 `DrawWater`),支持 30 帧循环水面贴图、相机空间水面变换矩阵与基于高度差的顶点 Alpha 混合。 + - **天空盒与动态云层(Skybox & Clouds)**:移植 `extension/src/platform/EterLib/SkyBox.cpp`(`CSkyObjectQuad`、`CSkyObject`、`CSkyBox::Render`、`CSkyBox::RenderCloud`),并在 `RecordingDevice.cpp` 中支持 Stage 0 二维纹理矩阵变换(`D3DTTFF_COUNT2`)与 `D3DTOP_MODULATEINVALPHA_ADDCOLOR`(19)云层着色模式。 + - **SpeedTree 森林与树木(SpeedTreeLib)**:实现 `SpeedTreeForest.cpp`、`SpeedTreeForestDirectX8.cpp` 与 `SpeedTreeWrapper.cpp`,读取 40250 `.spt` 属性与 `"TreeSize"` 参数生成带树皮贴图与树叶交叉十字网格的 Z-up 顶点/索引缓冲并参与渲染。 + - **多级纹理高光(Specular Sphere-Map)与完整混合因子**:在 `PythonApplication.cpp` 中实现 `CGrannyMaterial::CreateSphereMap` 与 `TranslateSpecularMatrix`,在 `native.vert` / `native.frag` 中实现球面环境映射 UV 生成与 `D3DTOP_MODULATEALPHA_ADDCOLOR`(18)高光叠加,并补齐完整 D3D8 `src_blend` / `dest_blend` 因子映射。 + - **修复退出纯虚函数崩溃**:移除基类析构函数 `CSkyObject::~CSkyObject()` 对纯虚函数 `Destroy()` 的调用,以及 `CSpeedTreeForest::~CSpeedTreeForest()` 对纯虚函数 `Clear()` 的调用(移入派生类 `CSpeedTreeForestDirectX8::~CSpeedTreeForestDirectX8()`),解决客户端退出时 `libc++abi: Pure virtual function called!` 崩溃。 +3. **音频播放接入(SDL3 Audio + AudioToolbox)**: + - 在 `native_render/main.cpp` 中实现 `NativeAudioEngine`,每帧排空 `DrainAudioCommands()`,通过 macOS `AudioToolbox`(`AudioFileOpenWithCallbacks` + `ExtAudioFileRead`)直接从内存解包 `.wav` / `.mp3` 为 44.1 kHz 双声道浮点 PCM,送入 `SDL_AudioStream` 完成 2D/3D 音效混音与 BGM 淡入淡出循环播放。 +4. **完整键盘/IME 文本输入与原生软件光标**: + - 补齐 `F1`–`F12`、`Tab`、`Backspace`、`Return`、`Delete`、`Shift`/`Ctrl`/`Alt`、方向键、数字小键盘及符号键的 `SDL_Scancode -> DIK_*` 与 `VK_*`(`PythonBoot::UIIMEKeyDown`)映射,接入 `SDL_EVENT_TEXT_INPUT` 转发 UTF-8 字符至 `PythonBoot::UIChar`。 + - 启用 40250 `CURSOR_MODE_SOFTWARE`(`OnMouseUpdate` / `OnMouseRender`),当游戏软件光标激活时自动隐藏系统光标并渲染 40250 原生彩色游戏光标。 +5. **Release (`-O3`) 构建与实测对比(Apple M4,180 帧,`--no-vsync`)**: + +| 场景与构建配置(Apple M4,180 帧) | 3D Draw | GPU 蒙皮 Draw | UI 批次 / 四边形 | 顶点数 / 索引数 | **`game_update_ms` (逻辑+UI+音频)** | **稳态 `steady_prepare_ms`** | **硬件 `gpu_ms`** | **`mean_frame_ms`** | +| :--- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| **完整场景 1 怪 · Debug 构建** | 97 | 10 | 32 / 221 | 225,447 / 91,257 | 5.82 ms | 0.62 ms | 0.44 ms | 8.27 ms (~121 FPS) | +| **完整场景 1 怪 · Release (`-O3`) 构建** | 96 | 10 | 32 / 221 | 225,443 / 91,251 | **1.36 ms** (**4.3x 加速**) | **0.79 ms** | **0.67 ms** | **8.33 ms** (~120 FPS) | +| **完整场景 64 怪 · Debug 构建** | 140 | 52 | 32 / 1,598 | 254,095 / 207,687 | 9.89 ms | 0.87 ms | 0.77 ms | 13.45 ms (~74 FPS) | +| **完整场景 64 怪 · Release (`-O3`) 构建** | **140** | **52** | **32 / 1,598** | **254,095 / 207,687** | **1.62 ms** (**6.1x 加速**) | **0.77 ms** | **1.00 ms** | **8.24 ms** (**~121 FPS**) | +| **完整场景 64 怪 `.mtdr` 回放 · Release (`-O3`)** | 140 | 52 | 31 / 1,597 | 254,095 / 207,687 | — | **0.19 ms** | **1.02 ms** | 8.62 ms | diff --git a/extension/src/metin2_world.cpp b/extension/src/metin2_world.cpp index 38b8fae4..21e2675b 100644 --- a/extension/src/metin2_world.cpp +++ b/extension/src/metin2_world.cpp @@ -369,6 +369,9 @@ const Metin2World::Chunk *Metin2World::chunk_at(int tx, int ty) const { } bool Metin2World::build_chunk(int tx, int ty) { + const bool profile_map = std::getenv("MT_PROFILE_MAP") != nullptr; + const auto ticks = [] { return Time::get_singleton()->get_ticks_usec(); }; + const double profile_start = profile_map ? ticks() : 0; const std::string dir = std::string(map_dir().utf8().get_data()) + "/" + fmt::m2coord::tile_dir(tx, ty); auto hm = std::make_shared(); @@ -378,12 +381,15 @@ bool Metin2World::build_chunk(int tx, int ty) { ++chunks_failed; return false; } + const double profile_height = profile_map ? ticks() : 0; auto am = std::make_shared(); if (!fmt::load_attr_map(dir + "/attr.atr", *am, &err)) am.reset(); // 非致命:attr 缺失 -> 该区块无阻挡 + const double profile_attr = profile_map ? ticks() : 0; fmt::TerrainMesh tmesh; fmt::build_terrain_mesh(*hm, tx, ty, setting.height_scale, tmesh); + const double profile_terrain = profile_map ? ticks() : 0; Ref mat; { @@ -394,19 +400,26 @@ bool Metin2World::build_chunk(int tx, int ty) { } bool splatted = false; + double profile_tile_ms = 0, profile_alpha_ms = 0, profile_material_ms = 0; if (splat_ready && resolver) { fmt::TileMap tile; std::string e2; + const double tile_start = profile_map ? ticks() : 0; if (fmt::load_tile_map(dir + "/tile.raw", tile, &e2)) { + if (profile_map) profile_tile_ms = (ticks() - tile_start) / 1000.0; fmt::SplatSet ss; + const double alpha_start = profile_map ? ticks() : 0; fmt::build_splat(tile, texture_set.runtime_count(), ss); + if (profile_map) profile_alpha_ms = (ticks() - alpha_start) / 1000.0; if (!ss.layers.empty()) { String smpath; String sm = String(dir.c_str()) + "/shadowmap.dds"; if (mtgodot::file_exists(sm)) smpath = sm; + const double material_start = profile_map ? ticks() : 0; Ref tm = build_chunk_terrain_material( ss, texture_set, *resolver, smpath); + if (profile_map) profile_material_ms = (ticks() - material_start) / 1000.0; if (tm.is_valid()) { mat = tm; splatted = true; @@ -416,6 +429,7 @@ bool Metin2World::build_chunk(int tx, int ty) { } if (splatted) ++chunks_splatted; + const double profile_splat = profile_map ? ticks() : 0; // 该区块的场景根 —— terrain / water / 对象 / 树都挂它下面,卸载 = free 它 Node3D *croot = memnew(Node3D); @@ -483,6 +497,7 @@ bool Metin2World::build_chunk(int tx, int ty) { } } ++chunks_built; + const double profile_godot_mesh = profile_map ? ticks() : 0; // 地形碰撞(W1 item 6):HeightMapShape3D,129×129,格距 = CELL_M(缩放承载) if (collision_enabled) { @@ -545,6 +560,18 @@ bool Metin2World::build_chunk(int tx, int ty) { objects_placed += ck.objects; trees_placed += ck.trees; chunks.push_back(std::move(ck)); + if (profile_map) { + const double profile_end = ticks(); + UtilityFunctions::print(vformat("MAP_PROFILE %d,%d height=%.3f attr=%.3f terrain_cpu=%.3f splat=%.3f tile=%.3f alpha=%.3f material=%.3f godot_mesh=%.3f rest=%.3f total=%.3f", + tx, ty, (profile_height - profile_start) / 1000.0, + (profile_attr - profile_height) / 1000.0, + (profile_terrain - profile_attr) / 1000.0, + (profile_splat - profile_terrain) / 1000.0, + profile_tile_ms, profile_alpha_ms, profile_material_ms, + (profile_godot_mesh - profile_splat) / 1000.0, + (profile_end - profile_godot_mesh) / 1000.0, + (profile_end - profile_start) / 1000.0)); + } return true; } diff --git a/extension/src/platform/EterGrnLib/GrannyRuntime.cpp b/extension/src/platform/EterGrnLib/GrannyRuntime.cpp index fc1d26ab..83083030 100644 --- a/extension/src/platform/EterGrnLib/GrannyRuntime.cpp +++ b/extension/src/platform/EterGrnLib/GrannyRuntime.cpp @@ -13,15 +13,21 @@ #include #include "granny.h" +#include "../EterLib/RenderCommands3D.h" #include #include +#include #include +#include #include +#include #include #include +#include #include + namespace { using Mat4 = gr2::Mat4; @@ -815,16 +821,179 @@ int read_influences(const Layout& l, int m, const uint8_t* v, float* out, bool w } return n; } +struct StaticSkinMesh +{ + std::uint64_t source_mesh_key = 0; + std::uint32_t vertex_count = 0; + std::uint32_t out_stride = 0; + std::uint32_t bone_count = 1; + std::vector bind_vertices; + std::vector bone_indices; // vertex_count * 4 + std::vector bone_weights; // vertex_count * 4 +}; + +struct DestSkinSlice +{ + const std::uint8_t* dest_begin = nullptr; + const std::uint8_t* dest_end = nullptr; + std::uint32_t stride = 0; + std::uint32_t vertex_count = 0; + const StaticSkinMesh* source = nullptr; + std::vector bone_matrices; // bone_count * 16 +}; + +std::mutex g_skin_mutex; +int g_gpu_skinning_override = -1; +std::unordered_map> g_static_skin_meshes; +std::map g_dest_skin_slices; } // namespace +void SetGpuSkinningEnabled(bool enabled) +{ + std::lock_guard lock(g_skin_mutex); + g_gpu_skinning_override = enabled ? 1 : 0; + if (!enabled) + g_dest_skin_slices.clear(); +} + +bool IsGpuSkinningEnabled() +{ + if (g_gpu_skinning_override >= 0) + return g_gpu_skinning_override != 0; + static const bool env_on = [] { + const char* v = std::getenv("MT_GPU_SKINNING"); + return v && (*v == '1' || *v == 't' || *v == 'T' || *v == 'y' || *v == 'Y'); + }(); + return env_on; +} + +bool LookupGpuSkinSubrange(const void* vertex_buffer_base, std::uint32_t stride, + std::uint32_t lo_vertex, std::uint32_t hi_vertex, + GpuSkinSubrangeView* out_view) +{ + if (!IsGpuSkinningEnabled() || !vertex_buffer_base || !stride || !out_view || hi_vertex < lo_vertex) + return false; + std::lock_guard lock(g_skin_mutex); + const auto* base = static_cast(vertex_buffer_base); + const auto* q_begin = base + std::size_t(lo_vertex) * stride; + const auto* q_end = base + std::size_t(hi_vertex + 1) * stride; + auto it = g_dest_skin_slices.upper_bound(q_begin); + if (it == g_dest_skin_slices.begin()) + return false; + --it; + const DestSkinSlice& slice = it->second; + if (!slice.source || slice.stride != stride || q_begin < slice.dest_begin || q_end > slice.dest_end || slice.dest_begin < base) + return false; + const std::size_t byte_offset = static_cast(slice.dest_begin - base); + if (byte_offset % stride != 0) + return false; + out_view->source_mesh_key = slice.source->source_mesh_key; + out_view->mesh_base_vertex = static_cast(byte_offset / stride); + out_view->mesh_vertex_count = slice.vertex_count; + out_view->bone_indices = slice.source->bone_indices.data(); + out_view->bone_weights = slice.source->bone_weights.data(); + out_view->bone_matrices = slice.bone_matrices.data(); + out_view->bone_count = slice.source->bone_count; + return true; +} + void GrannyDeformVertices(granny_mesh_deformer const* Deformer, granny_int32x const* MatrixIndices, granny_real32 const* MatrixBuffer4x4, granny_int32x VertexCount, void const* SourceVertices, void* DestVertices) { - if (!Deformer || !MatrixBuffer4x4 || !SourceVertices || !DestVertices) + if (!Deformer || !MatrixBuffer4x4 || !SourceVertices || !DestVertices || VertexCount <= 0) return; const granny_mesh_deformer& d = *Deformer; const bool normals = d.type != GrannyDeformPosition && d.in_normal >= 0 && d.out_normal >= 0; + + if (IsGpuSkinningEnabled()) + { + std::lock_guard lock(g_skin_mutex); + auto& mesh_ptr = g_static_skin_meshes[SourceVertices]; + if (!mesh_ptr || mesh_ptr->vertex_count != std::uint32_t(VertexCount) || mesh_ptr->out_stride != std::uint32_t(d.out.size)) + { + mesh_ptr = std::make_unique(); + mesh_ptr->source_mesh_key = static_cast(reinterpret_cast(SourceVertices)); + mesh_ptr->vertex_count = static_cast(VertexCount); + mesh_ptr->out_stride = static_cast(d.out.size); + mesh_ptr->bind_vertices.resize(std::size_t(VertexCount) * d.out.size, 0); + mesh_ptr->bone_indices.resize(std::size_t(VertexCount) * 4, 0); + mesh_ptr->bone_weights.resize(std::size_t(VertexCount) * 4, 0.0f); + int max_local_bone = 0; + for (int v = 0; v < VertexCount; ++v) + { + const uint8_t* src = static_cast(SourceVertices) + std::size_t(v) * d.in.size; + uint8_t* dst = mesh_ptr->bind_vertices.data() + std::size_t(v) * d.out.size; + run_conversion(d.tail, src, dst); + float p[3] = {0, 0, 0}, nrm[3] = {0, 0, 0}; + read_vec(d.in, d.in_position, src, p, 3); + write_vec(d.out, d.out_position, dst, p, 3); + if (normals) + { + read_vec(d.in, d.in_normal, src, nrm, 3); + write_vec(d.out, d.out_normal, dst, nrm, 3); + } + float weights[4] = {1, 0, 0, 0}, indices[4] = {0, 0, 0, 0}; + int wn = 1, in = 1; + if (d.in_weights >= 0) + wn = read_influences(d.in, d.in_weights, src, weights, true); + if (d.in_indices >= 0) + in = read_influences(d.in, d.in_indices, src, indices, false); + const int n = std::min(wn, in); + for (int k = 0; k < n; ++k) + { + const float w = weights[k]; + if (w <= 0.0f) + continue; + const int local = std::max(0, std::min(127, int(indices[k]))); + mesh_ptr->bone_indices[std::size_t(v) * 4 + k] = static_cast(local); + mesh_ptr->bone_weights[std::size_t(v) * 4 + k] = w; + if (local > max_local_bone) + max_local_bone = local; + } + } + mesh_ptr->bone_count = static_cast(max_local_bone + 1); + } + + const StaticSkinMesh* mesh = mesh_ptr.get(); + const auto* dest_begin = static_cast(DestVertices); + const auto* dest_end = dest_begin + mesh->bind_vertices.size(); + + // Remove any stale overlapping slice from a previous model using the same buffer. + auto ov = g_dest_skin_slices.lower_bound(dest_begin); + if (ov != g_dest_skin_slices.begin()) + { + auto prev = std::prev(ov); + if (prev->second.dest_end > dest_begin) + ov = prev; + } + while (ov != g_dest_skin_slices.end() && ov->second.dest_begin < dest_end) + { + if (ov->first != dest_begin) + ov = g_dest_skin_slices.erase(ov); + else + ++ov; + } + + DestSkinSlice& slice = g_dest_skin_slices[dest_begin]; + if (slice.source != mesh || slice.vertex_count != mesh->vertex_count) + { + std::memcpy(DestVertices, mesh->bind_vertices.data(), mesh->bind_vertices.size()); + slice.dest_begin = dest_begin; + slice.dest_end = dest_end; + slice.stride = mesh->out_stride; + slice.vertex_count = mesh->vertex_count; + slice.source = mesh; + } + slice.bone_matrices.resize(std::size_t(mesh->bone_count) * 16); + for (std::uint32_t b = 0; b < mesh->bone_count; ++b) + { + const int bone = MatrixIndices ? MatrixIndices[b] : int(b); + std::memcpy(&slice.bone_matrices[std::size_t(b) * 16], MatrixBuffer4x4 + std::size_t(bone) * 16, 16 * sizeof(float)); + } + return; + } + for (int v = 0; v < VertexCount; ++v) { const uint8_t* src = static_cast(SourceVertices) + size_t(v) * d.in.size; @@ -863,6 +1032,7 @@ void GrannyDeformVertices(granny_mesh_deformer const* Deformer, granny_int32x co } } + // ── Model instances and controls ─────────────────────────────────────────────────────────────── struct granny_model_instance diff --git a/extension/src/platform/EterLib/CpuBuffer.h b/extension/src/platform/EterLib/CpuBuffer.h index 767a5f24..1ba43ca1 100644 --- a/extension/src/platform/EterLib/CpuBuffer.h +++ b/extension/src/platform/EterLib/CpuBuffer.h @@ -5,10 +5,15 @@ #include "EterLib/StdAfx.h" #include +#include #include +inline std::atomic mt_next_cpu_buffer_id{1}; + struct MtCpuVertexBuffer : IDirect3DVertexBuffer8 { + const std::uint64_t id = mt_next_cpu_buffer_id.fetch_add(1, std::memory_order_relaxed); + std::uint64_t revision = 0; std::vector bytes; DWORD fvf = 0; @@ -19,11 +24,13 @@ struct MtCpuVertexBuffer : IDirect3DVertexBuffer8 *data = bytes.data() + offset; return S_OK; } - HRESULT Unlock() override { return S_OK; } + HRESULT Unlock() override { ++revision; return S_OK; } }; struct MtCpuIndexBuffer : IDirect3DIndexBuffer8 { + const std::uint64_t id = mt_next_cpu_buffer_id.fetch_add(1, std::memory_order_relaxed); + std::uint64_t revision = 0; std::vector bytes; D3DFORMAT format = D3DFMT_INDEX16; @@ -34,7 +41,7 @@ struct MtCpuIndexBuffer : IDirect3DIndexBuffer8 *data = bytes.data() + offset; return S_OK; } - HRESULT Unlock() override { return S_OK; } + HRESULT Unlock() override { ++revision; return S_OK; } }; // D3DXGetFVFVertexSize. diff --git a/extension/src/platform/EterLib/CullingManager.cpp b/extension/src/platform/EterLib/CullingManager.cpp deleted file mode 100644 index 532a7b8d..00000000 --- a/extension/src/platform/EterLib/CullingManager.cpp +++ /dev/null @@ -1,79 +0,0 @@ -// Platform skeleton for EterLib/CullingManager.h (40250 EterLib/CullingManager.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. -#include "EterLib/StdAfx.h" -#include "EterLib/CullingManager.h" - -#include "../PlatformStub.h" - -CCullingManager::CCullingManager() -{ - MT_PLATFORM_STUB(); -} - -CCullingManager::~CCullingManager() -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::RayTraceCallback(const Vector3d &, const Vector3d &, float, const Vector3d &, SpherePack *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::VisibilityCallback(const Frustum &, SpherePack *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::RangeTestCallback(const Vector3d &, float, SpherePack *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::Reset() -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::Update() -> void -{ - MT_PLATFORM_STUB(); -} - -// 40250 CullingManager.cpp:105. RenderGame calls it after SetPerspective; it hands the camera's view and -// the projection to CStateManager for the scene. -// PORT: BuildViewFrustum and m_Factory->FrustumTest follow in 40250; the sphere-pack culling (SphereLib) -// is not ported, so every registered object stays visible. -void CCullingManager::Process() -{ - //DWORD time = ELTimer_GetMSec(); - //Frustum f; - UpdateViewMatrix(); - UpdateProjMatrix(); -} - -auto CCullingManager::FindRange(const Vector3d &, float) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::FindRay(const Vector3d &, const Vector3d &) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::FindRayDistance(const Vector3d &, const Vector3d &, float) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CCullingManager::Register(CGraphicObjectInstance *) -> CullingHandle -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto CCullingManager::Unregister(CullingHandle) -> void -{ - MT_PLATFORM_STUB(); -} diff --git a/extension/src/platform/EterLib/GrpDevice.cpp b/extension/src/platform/EterLib/GrpDevice.cpp index 15cb7eec..7e913cf6 100644 --- a/extension/src/platform/EterLib/GrpDevice.cpp +++ b/extension/src/platform/EterLib/GrpDevice.cpp @@ -98,6 +98,8 @@ int CGraphicDevice::Create(HWND hWnd, int iHres, int iVres, bool Windowed, int / D3DXMatrixIdentity(&ms_matProj); D3DXMatrixIdentity(&ms_matInverseView); D3DXMatrixIdentity(&ms_matInverseViewYAxis); + D3DXMatrixIdentity(&ms_matWorld); + D3DXMatrixIdentity(&ms_matWorldView); D3DXMatrixIdentity(&ms_matScreen0); D3DXMatrixIdentity(&ms_matScreen1); D3DXMatrixIdentity(&ms_matScreen2); diff --git a/extension/src/platform/EterLib/GrpImageTexture.cpp b/extension/src/platform/EterLib/GrpImageTexture.cpp index 953fcdfe..e2370873 100644 --- a/extension/src/platform/EterLib/GrpImageTexture.cpp +++ b/extension/src/platform/EterLib/GrpImageTexture.cpp @@ -137,16 +137,22 @@ std::string UIRenderTextureNameFromHandle(const IDirect3DBaseTexture8* handle) { if (!handle) return {}; const auto* texture = static_cast(handle); - std::lock_guard lock(g_memory_mutex); - if (const MemoryTexture* memory = live_memory_texture(texture)) - return "mem:" + std::to_string(memory->id) + "@" + std::to_string(memory->revision); - const auto* file = static_cast(texture); - return g_file_textures.count(file) ? file->name : std::string(); + { + std::lock_guard lock(g_memory_mutex); + if (const MemoryTexture* memory = live_memory_texture(texture)) + return "mem:" + std::to_string(memory->id) + "@" + std::to_string(memory->revision); + const auto* file = static_cast(texture); + if (g_file_textures.count(file)) + return file->name; + } + return MtCpuTextureNameFromHandle(handle); } bool UIRenderMemoryTexture(const std::string& name, UIMemoryTexture* out) { if (name.compare(0, 4, "mem:") != 0 || !out) return false; + if (name.compare(0, 8, "mem:cpu_") == 0) + return MtCpuMemoryTexture(name, out); const unsigned id = unsigned(std::strtoul(name.c_str() + 4, nullptr, 10)); std::lock_guard lock(g_memory_mutex); auto it = g_memory.find(id); diff --git a/extension/src/platform/EterLib/GrpIndexBuffer.cpp b/extension/src/platform/EterLib/GrpIndexBuffer.cpp index 548ffaea..525db706 100644 --- a/extension/src/platform/EterLib/GrpIndexBuffer.cpp +++ b/extension/src/platform/EterLib/GrpIndexBuffer.cpp @@ -34,6 +34,7 @@ bool CGraphicIndexBuffer::Lock(void** pretIndices) const void CGraphicIndexBuffer::Unlock() const { assert(m_lpd3dIdxBuf!=NULL); + static_cast(m_lpd3dIdxBuf)->revision++; } bool CGraphicIndexBuffer::Lock(void** pretIndices) @@ -47,6 +48,7 @@ bool CGraphicIndexBuffer::Lock(void** pretIndices) void CGraphicIndexBuffer::Unlock() { assert(m_lpd3dIdxBuf!=NULL); + static_cast(m_lpd3dIdxBuf)->revision++; } bool CGraphicIndexBuffer::Copy(int bufSize, const void* srcIndices) @@ -54,6 +56,7 @@ bool CGraphicIndexBuffer::Copy(int bufSize, const void* srcIndices) assert(m_lpd3dIdxBuf!=NULL); memcpy(index_bytes(m_lpd3dIdxBuf), srcIndices, bufSize); + static_cast(m_lpd3dIdxBuf)->revision++; return true; } @@ -73,6 +76,7 @@ bool CGraphicIndexBuffer::Create(int faceCount, TFace* faces) dstIndices[1]=curFace->indices[1]; dstIndices[2]=curFace->indices[2]; } + static_cast(m_lpd3dIdxBuf)->revision++; return true; } diff --git a/extension/src/platform/EterLib/GrpScreen.cpp b/extension/src/platform/EterLib/GrpScreen.cpp index d681f198..d37e219f 100644 --- a/extension/src/platform/EterLib/GrpScreen.cpp +++ b/extension/src/platform/EterLib/GrpScreen.cpp @@ -3,6 +3,7 @@ #include "EterLib/StdAfx.h" #include "EterLib/GrpScreen.h" #include "EterLib/StateManager.h" +#include "EterLib/Camera.h" #include "../PlatformStub.h" #include "UIRenderCommands.h" @@ -285,27 +286,75 @@ void CScreen::SetCursorPosition(int x, int y, int hres, int vres) ms_Ray.SetDirection(-ms_vtPickRayDir, 51200.0f); } -auto CScreen::GetCursorPosition(float *, float *, float *) -> bool +auto CScreen::GetCursorPosition(float * px, float * py, float * pz) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + if (!GetCursorXYPosition(px, py)) return false; + if (!GetCursorZPosition(pz)) return false; + return true; } -auto CScreen::GetCursorXYPosition(float *, float *) -> bool +auto CScreen::GetCursorXYPosition(float * px, float * py) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + D3DXVECTOR3 v3Eye = CCameraManager::Instance().GetCurrentCamera()->GetEye(); + + TPosition posVertices[4]; + posVertices[0] = TPosition(v3Eye.x - 90000000.0f, v3Eye.y + 90000000.0f, 0.0f); + posVertices[1] = TPosition(v3Eye.x - 90000000.0f, v3Eye.y - 90000000.0f, 0.0f); + posVertices[2] = TPosition(v3Eye.x + 90000000.0f, v3Eye.y + 90000000.0f, 0.0f); + posVertices[3] = TPosition(v3Eye.x + 90000000.0f, v3Eye.y - 90000000.0f, 0.0f); + + static const WORD sc_awFillRectIndices[6] = { 0, 2, 1, 2, 3, 1, }; + + float u, v, t; + for (int i = 0; i < 2; ++i) + { + if (IntersectTriangle(ms_vtPickRayOrig, ms_vtPickRayDir, + posVertices[sc_awFillRectIndices[i * 3]], + posVertices[sc_awFillRectIndices[i * 3 + 1]], + posVertices[sc_awFillRectIndices[i * 3 + 2]], + &u, &v, &t)) + { + *px = u; + *py = v; + return true; + } + } + return false; } -auto CScreen::GetCursorZPosition(float *) -> bool +auto CScreen::GetCursorZPosition(float * pz) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + D3DXVECTOR3 v3Eye = CCameraManager::Instance().GetCurrentCamera()->GetEye(); + + TPosition posVertices[4]; + posVertices[0] = TPosition(v3Eye.x - 90000000.0f, 0.0f, v3Eye.z + 90000000.0f); + posVertices[1] = TPosition(v3Eye.x - 90000000.0f, 0.0f, v3Eye.z - 90000000.0f); + posVertices[2] = TPosition(v3Eye.x + 90000000.0f, 0.0f, v3Eye.z + 90000000.0f); + posVertices[3] = TPosition(v3Eye.x + 90000000.0f, 0.0f, v3Eye.z - 90000000.0f); + + static const WORD sc_awFillRectIndices[6] = { 0, 2, 1, 2, 3, 1, }; + + float u, v, t; + for (int i = 0; i < 2; ++i) + { + if (IntersectTriangle(ms_vtPickRayOrig, ms_vtPickRayDir, + posVertices[sc_awFillRectIndices[i * 3]], + posVertices[sc_awFillRectIndices[i * 3 + 1]], + posVertices[sc_awFillRectIndices[i * 3 + 2]], + &u, &v, &t)) + { + *pz = t; + return true; + } + } + return false; } -auto CScreen::GetPickingPosition(float, float *, float *, float *) -> void +auto CScreen::GetPickingPosition(float t, float * x, float * y, float * z) -> void { - MT_PLATFORM_STUB(); + *x = ms_vtPickRayOrig.x + ms_vtPickRayDir.x * t; + *y = ms_vtPickRayOrig.y + ms_vtPickRayDir.y * t; + *z = ms_vtPickRayOrig.z + ms_vtPickRayDir.z * t; } // 40250 GrpScreen.cpp:749-779. @@ -313,7 +362,7 @@ void CScreen::ProjectPosition(float x, float y, float z, float * pfX, float * pf { D3DXVECTOR3 Input(x, y, z); D3DXVECTOR3 Output; - D3DXVec3Project(&Output, &Input, &ms_Viewport, &ms_matProj, &ms_matView, &ms_matWorld); + D3DXVec3Project(&Output, &Input, &ms_Viewport, &ms_matProj, &ms_matView, &ms_matIdentity); *pfX = Output.x; *pfY = Output.y; @@ -323,7 +372,7 @@ void CScreen::ProjectPosition(float x, float y, float z, float * pfX, float * pf { D3DXVECTOR3 Input(x, y, z); D3DXVECTOR3 Output; - D3DXVec3Project(&Output, &Input, &ms_Viewport, &ms_matProj, &ms_matView, &ms_matWorld); + D3DXVec3Project(&Output, &Input, &ms_Viewport, &ms_matProj, &ms_matView, &ms_matIdentity); *pfX = Output.x; *pfY = Output.y; @@ -334,7 +383,7 @@ void CScreen::UnprojectPosition(float x, float y, float z, float * pfX, float * { D3DXVECTOR3 Input(x, y, z); D3DXVECTOR3 Output; - D3DXVec3Unproject(&Output, &Input, &ms_Viewport, &ms_matProj, &ms_matView, &ms_matWorld); + D3DXVec3Unproject(&Output, &Input, &ms_Viewport, &ms_matProj, &ms_matView, &ms_matIdentity); *pfX = Output.x; *pfY = Output.y; @@ -355,12 +404,24 @@ auto CScreen::RestoreDevice() -> BOOL auto CScreen::BuildViewFrustum() -> void { - MT_PLATFORM_STUB(); + CCamera* pkCamera = CCameraManager::Instance().GetCurrentCamera(); + if (!pkCamera) + return; + const D3DXVECTOR3& c_rv3Eye = pkCamera->GetEye(); + const D3DXVECTOR3& c_rv3View = pkCamera->GetView(); + auto vv = ms_matView * ms_matProj; + ms_frustum.BuildViewFrustum2( + vv, + ms_fNearY, + ms_fFarY, + ms_fFieldOfView, + ms_fAspect, + c_rv3Eye, c_rv3View); } auto CScreen::Identity() -> void { - MT_PLATFORM_STUB(); + STATEMANAGER.SetTransform(D3DTS_WORLD, &ms_matIdentity); } decltype(CScreen::ms_diffuseColor) CScreen::ms_diffuseColor{}; diff --git a/extension/src/platform/EterLib/GrpVertexBuffer.cpp b/extension/src/platform/EterLib/GrpVertexBuffer.cpp index ef09f9ad..26237b4f 100644 --- a/extension/src/platform/EterLib/GrpVertexBuffer.cpp +++ b/extension/src/platform/EterLib/GrpVertexBuffer.cpp @@ -49,6 +49,7 @@ bool CGraphicVertexBuffer::Unlock() const { if (!m_lpd3dVB) return false; + static_cast(m_lpd3dVB)->revision++; return true; } @@ -83,6 +84,7 @@ bool CGraphicVertexBuffer::Unlock() { if (!m_lpd3dVB) return false; + static_cast(m_lpd3dVB)->revision++; return true; } diff --git a/extension/src/platform/EterLib/Input.cpp b/extension/src/platform/EterLib/Input.cpp index 1b505429..93f30072 100644 --- a/extension/src/platform/EterLib/Input.cpp +++ b/extension/src/platform/EterLib/Input.cpp @@ -35,34 +35,43 @@ CInputKeyboard::~CInputKeyboard() auto CInputKeyboard::InitializeKeyboard(HWND) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + ResetKeyboard(); + return true; } auto CInputKeyboard::UpdateKeyboard() -> void { - MT_PLATFORM_STUB(); } auto CInputKeyboard::ResetKeyboard() -> void { - MT_PLATFORM_STUB(); + memset(ms_bPressedKey, 0, sizeof(ms_bPressedKey)); + memset(ms_diks, 0, sizeof(ms_diks)); } -auto CInputKeyboard::IsPressed(int) -> bool +auto CInputKeyboard::IsPressed(int iIndex) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + if (iIndex >= 0 && iIndex < 256) + return (ms_diks[iIndex] & 0x80) != 0; + return false; } -auto CInputKeyboard::KeyDown(int) -> void +auto CInputKeyboard::KeyDown(int iIndex) -> void { - MT_PLATFORM_STUB(); + if (iIndex >= 0 && iIndex < 256) + { + ms_bPressedKey[iIndex] = true; + ms_diks[iIndex] = (char)0x80; + } } -auto CInputKeyboard::KeyUp(int) -> void +auto CInputKeyboard::KeyUp(int iIndex) -> void { - MT_PLATFORM_STUB(); + if (iIndex >= 0 && iIndex < 256) + { + ms_bPressedKey[iIndex] = false; + ms_diks[iIndex] = 0; + } } decltype(CInputKeyboard::ms_lpKeyboard) CInputKeyboard::ms_lpKeyboard{}; diff --git a/extension/src/platform/EterLib/RecordingDevice.cpp b/extension/src/platform/EterLib/RecordingDevice.cpp index ad296d1c..4b4626c3 100644 --- a/extension/src/platform/EterLib/RecordingDevice.cpp +++ b/extension/src/platform/EterLib/RecordingDevice.cpp @@ -5,12 +5,17 @@ #include #include +#include #include +#include #include +#include namespace { std::mutex g_draws_mutex; std::vector g_draws; +int g_native_terrain_override = -1; + struct VertexLayout { unsigned stride = 0; @@ -69,9 +74,8 @@ int format_bytes(D3DFORMAT format) } // PORT: device-created resources live in CPU memory with D3D reference counting. Textures made by -// CreateTexture (CTerrain's splat alpha maps and attribute marks, CSnowEnvironment's blur targets) -// keep their mip levels as bytes; nothing samples them because terrain and offscreen passes are -// drawn by the Godot side. +// CreateTexture resources keep their mip levels as bytes. Terrain alpha maps and the character +// shadow render target are sampled by the native renderer through MtCpuMemoryTexture. struct CpuTexture; struct CpuSurface final : IDirect3DSurface8 @@ -89,13 +93,32 @@ struct CpuSurface final : IDirect3DSurface8 HRESULT UnlockRect() override { return S_OK; } }; +struct CpuTexture; +std::mutex g_cpu_tex_mutex; +std::unordered_map g_cpu_textures; +unsigned g_next_cpu_tex_id = 1; + struct CpuTexture final : IDirect3DTexture8 { struct Level { D3DSURFACE_DESC desc; std::vector bytes; }; + unsigned id = 0; + std::uint32_t revision = 1; ULONG refs = 1; std::vector levels; + CpuTexture() + { + std::lock_guard lock(g_cpu_tex_mutex); + id = g_next_cpu_tex_id++; + g_cpu_textures[id] = this; + } + ~CpuTexture() + { + std::lock_guard lock(g_cpu_tex_mutex); + g_cpu_textures.erase(id); + } + ULONG AddRef() override { return ++refs; } ULONG Release() override { @@ -132,7 +155,14 @@ struct CpuTexture final : IDirect3DTexture8 locked->pBits = levels[level].bytes.data(); return S_OK; } - HRESULT UnlockRect(UINT level) override { return level < levels.size() ? S_OK : E_FAIL; } + HRESULT UnlockRect(UINT level) override + { + if (level >= levels.size()) + return E_FAIL; + if (level == 0) + ++revision; + return S_OK; + } }; ULONG CpuSurface::Release() @@ -217,6 +247,22 @@ public: m_renderTarget->AddRef(); m_depthStencil = m_depthBuffer; m_depthStencil->AddRef(); + m_viewport = { 0, 0, (DWORD)width, (DWORD)height, 0.0f, 1.0f }; + m_renderStates[D3DRS_TEXTUREFACTOR] = 0xFFFFFFFF; + m_renderStates[D3DRS_ZENABLE] = TRUE; + m_renderStates[D3DRS_ZWRITEENABLE] = TRUE; + m_renderStates[D3DRS_CULLMODE] = D3DCULL_CCW; + m_renderStates[D3DRS_LIGHTING] = TRUE; + m_renderStates[D3DRS_SRCBLEND] = D3DBLEND_ONE; + m_renderStates[D3DRS_DESTBLEND] = D3DBLEND_ZERO; + m_renderStates[D3DRS_ALPHAFUNC] = D3DCMP_ALWAYS; + m_stageStates[0][D3DTSS_COLOROP] = D3DTOP_MODULATE; + m_stageStates[0][D3DTSS_COLORARG1] = D3DTA_TEXTURE; + m_stageStates[0][D3DTSS_COLORARG2] = D3DTA_CURRENT; + m_stageStates[0][D3DTSS_ALPHAOP] = D3DTOP_SELECTARG1; + m_stageStates[0][D3DTSS_ALPHAARG1] = D3DTA_TEXTURE; + m_stageStates[1][D3DTSS_COLOROP] = D3DTOP_DISABLE; + m_stageStates[1][D3DTSS_ALPHAOP] = D3DTOP_DISABLE; } ~RecordingDevice() override { @@ -254,7 +300,14 @@ public: } HRESULT GetViewport(D3DVIEWPORT8* viewport) override { - *viewport = { 0, 0, DWORD(m_width), DWORD(m_height), 0.0f, 1.0f }; + if (viewport) + *viewport = m_viewport; + return S_OK; + } + HRESULT SetViewport(const D3DVIEWPORT8* viewport) override + { + if (viewport) + m_viewport = *viewport; return S_OK; } // PORT: CPU memory has no texture budget; report the 64MB of a mid-range 2004 card. @@ -340,8 +393,8 @@ public: *out = m_depthStencil; return S_OK; } - // PORT: an offscreen target (CSnowEnvironment's blur pass) swallows its draws; only the back - // buffer's draws are recorded for the Godot renderer. + // Character shadow targets are rasterized below; other offscreen targets still need a + // separate effect-specific implementation (snow blur is disabled by default). HRESULT SetRenderTarget(IDirect3DSurface8* target, IDirect3DSurface8* depth) override { if (target) @@ -358,7 +411,28 @@ public: } return S_OK; } - HRESULT Clear(DWORD, const D3DRECT*, DWORD, D3DCOLOR, float, DWORD) override { return S_OK; } + HRESULT Clear(DWORD, const D3DRECT*, DWORD flags, D3DCOLOR color, float depth, DWORD) override + { + if (m_renderTarget == m_backBuffer) + return S_OK; + auto* surface = static_cast(m_renderTarget); + if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5) + return S_OK; + const size_t pixels = size_t(surface->desc.Width) * surface->desc.Height; + if (flags & D3DCLEAR_TARGET) + { + const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 | + (((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3)); + auto& bytes = surface->parent->levels[0].bytes; + bytes.resize(pixels * 2); + for (size_t i = 0; i < pixels; ++i) + std::memcpy(bytes.data() + i * 2, &rgb565, 2); + ++surface->parent->revision; + } + if (flags & D3DCLEAR_ZBUFFER) + m_offscreenDepth.assign(pixels, depth); + return S_OK; + } HRESULT SetRenderState(D3DRENDERSTATETYPE state, DWORD value) override { if (unsigned(state) < kRenderStates) @@ -406,7 +480,8 @@ public: std::vector indices(count); for (UINT i = 0; i < count; ++i) indices[i] = startVertex + i; - record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices); + record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices, + m_stream, nullptr, startVertex); return S_OK; } HRESULT DrawIndexedPrimitive(D3DPRIMITIVETYPE type, UINT, UINT, UINT startIndex, UINT primitiveCount) override @@ -417,7 +492,8 @@ public: if (!read_indices(m_indices->bytes.data(), m_indices->bytes.size(), m_indices->format, startIndex, index_count(type, primitiveCount), m_baseVertex, &indices)) return E_FAIL; - record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices); + record(type, primitiveCount, m_stream->bytes.data(), m_stream->bytes.size(), m_streamStride, indices, + m_stream, m_indices, startIndex); return S_OK; } HRESULT DrawPrimitiveUP(D3DPRIMITIVETYPE type, UINT primitiveCount, const void* vertices, UINT stride) override @@ -488,6 +564,92 @@ private: transform_point(&a[r * 4], b, &out[r * 4]); } + // The 40250 shadow pass renders a flat grey silhouette into an R5G6B5 target. + // Rasterizing that small target here keeps the D3D render-target sequence and lets + // the normal memory-texture path upload the result for terrain/object projection. + void rasterize_shadow(const Render3DDraw& draw) + { + auto* surface = static_cast(m_renderTarget); + if (!surface->parent || surface->level != 0 || surface->desc.Format != D3DFMT_R5G6B5 || draw.lines) + return; + const int width = int(surface->desc.Width), height = int(surface->desc.Height); + if (width <= 0 || height <= 0 || draw.indices.size() < 3) + return; + auto& bytes = surface->parent->levels[0].bytes; + if (bytes.size() < size_t(width) * height * 2) + return; + if (m_offscreenDepth.size() != size_t(width) * height) + m_offscreenDepth.assign(size_t(width) * height, 1.0f); + float worldView[16], mvp[16]; + multiply(draw.world, draw.view, worldView); + multiply(worldView, draw.proj, mvp); + struct Point { float x, y, z; bool valid; }; + std::vector points(draw.positions.size() / 3); + for (size_t i = 0; i < points.size(); ++i) + { + float p[4] = { draw.positions[i * 3], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f }; + if (!draw.bone_matrices.empty() && draw.bone_weights.size() >= (i + 1) * 4 && + draw.bone_indices.size() >= (i + 1) * 4) + { + float skinned[4] = {}; + float total = 0.0f; + for (int b = 0; b < 4; ++b) + { + const float weight = draw.bone_weights[i * 4 + b]; + const size_t bone = draw.bone_indices[i * 4 + b]; + if (weight <= 0.0f || (bone + 1) * 16 > draw.bone_matrices.size()) continue; + float transformed[4]; + transform_point(p, draw.bone_matrices.data() + bone * 16, transformed); + for (int c = 0; c < 4; ++c) skinned[c] += transformed[c] * weight; + total += weight; + } + if (total > 0.0f) { p[0] = skinned[0]; p[1] = skinned[1]; p[2] = skinned[2]; } + } + float clip[4]; + transform_point(p, mvp, clip); + const bool valid = clip[3] > 0.00001f; + const float invW = valid ? 1.0f / clip[3] : 0.0f; + points[i] = { draw.viewport[0] + (clip[0] * invW * 0.5f + 0.5f) * draw.viewport[2], + draw.viewport[1] + (0.5f - clip[1] * invW * 0.5f) * draw.viewport[3], + clip[2] * invW, valid }; + } + const std::uint32_t color = m_renderStates[D3DRS_TEXTUREFACTOR]; + const std::uint16_t rgb565 = std::uint16_t((((color >> 16) & 255u) >> 3) << 11 | + (((color >> 8) & 255u) >> 2) << 5 | ((color & 255u) >> 3)); + auto edge = [](const Point& a, const Point& b, float x, float y) { + return (x - a.x) * (b.y - a.y) - (y - a.y) * (b.x - a.x); + }; + for (size_t i = 0; i + 2 < draw.indices.size(); i += 3) + { + if (draw.indices[i] >= points.size() || draw.indices[i + 1] >= points.size() || + draw.indices[i + 2] >= points.size()) continue; + const Point& a = points[draw.indices[i]]; + const Point& b = points[draw.indices[i + 1]]; + const Point& c = points[draw.indices[i + 2]]; + if (!a.valid || !b.valid || !c.valid) continue; + const float area = edge(a, b, c.x, c.y); + if (std::fabs(area) < 0.00001f) continue; + const int x0 = std::max(0, int(std::floor(std::min({a.x, b.x, c.x})))); + const int y0 = std::max(0, int(std::floor(std::min({a.y, b.y, c.y})))); + const int x1 = std::min(width - 1, int(std::ceil(std::max({a.x, b.x, c.x})))); + const int y1 = std::min(height - 1, int(std::ceil(std::max({a.y, b.y, c.y})))); + for (int y = y0; y <= y1; ++y) + for (int x = x0; x <= x1; ++x) + { + const float px = float(x) + 0.5f, py = float(y) + 0.5f; + const float wa = edge(b, c, px, py) / area; + const float wb = edge(c, a, px, py) / area; + const float wc = 1.0f - wa - wb; + if (wa < 0.0f || wb < 0.0f || wc < 0.0f) continue; + const float z = wa * a.z + wb * b.z + wc * c.z; + const size_t pixel = size_t(y) * width + x; + if (z < 0.0f || z > 1.0f || z > m_offscreenDepth[pixel]) continue; + m_offscreenDepth[pixel] = z; + std::memcpy(bytes.data() + pixel * 2, &rgb565, 2); + } + } + } + // Each two-triangle quad of a UI-space draw becomes an image command: its screen corners through // world * view * projection, its stage-0 texture (or D3DTA_TFACTOR colour when stage 0 selects it) // and, when stage 1 generates coordinates from the camera-space position through D3DTS_TEXTURE1 @@ -585,9 +747,10 @@ private: } void record(D3DPRIMITIVETYPE type, UINT primitives, const uint8_t* vertices, size_t vertexBytes, UINT stride, - const std::vector& indices) + const std::vector& indices, const MtCpuVertexBuffer* source_vertex = nullptr, + const MtCpuIndexBuffer* source_index = nullptr, UINT source_first_index = 0) { - if (type == D3DPT_POINTLIST || indices.empty() || !vertices || m_renderTarget != m_backBuffer) + if (type == D3DPT_POINTLIST || indices.empty() || !vertices) return; VertexLayout layout = fvf_layout(m_fvf); if (!stride) @@ -602,7 +765,7 @@ private: // An orthographic projection over untransformed vertices is a UI-space draw (CPythonMiniMap's // terrain tiles under CPythonGraphic::SetOrtho2D): it joins the UI command stream in order. - if (!layout.rhw && m_transforms[D3DTS_PROJECTION][11] == 0.0f) + if (m_renderTarget == m_backBuffer && !layout.rhw && m_transforms[D3DTS_PROJECTION][11] == 0.0f) { record_ui_quads(type, vertices, vertexBytes, stride, layout, indices); return; @@ -614,6 +777,10 @@ private: std::memcpy(draw.proj, m_transforms[D3DTS_PROJECTION], sizeof(draw.proj)); draw.texture0 = UIRenderTextureNameFromHandle(m_textures[0]); draw.texture1 = UIRenderTextureNameFromHandle(m_textures[1]); + draw.viewport[0] = float(m_viewport.X); + draw.viewport[1] = float(m_viewport.Y); + draw.viewport[2] = float(m_viewport.Width); + draw.viewport[3] = float(m_viewport.Height); draw.pretransformed = layout.rhw; draw.lines = type == D3DPT_LINELIST || type == D3DPT_LINESTRIP; @@ -627,47 +794,221 @@ private: if ((size_t(hi) + 1) * stride > vertexBytes) return; const size_t count = size_t(hi - lo) + 1; - draw.positions.resize(count * 3); - if (layout.rhw) draw.rhw.resize(count); - if (layout.normal >= 0) draw.normals.resize(count * 3); - if (layout.uv0 >= 0) draw.uv0.resize(count * 2); - if (layout.uv1 >= 0) draw.uv1.resize(count * 2); - if (layout.diffuse >= 0) draw.diffuse.resize(count); - for (size_t i = 0; i < count; ++i) + const bool texgen_cam_pos0 = layout.uv0 < 0 && m_textures[0] != nullptr && + (m_stageStates[0][D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == D3DTSS_TCI_CAMERASPACEPOSITION && + m_stageStates[0][D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2; + const bool texgen_cam_pos1 = layout.uv1 < 0 && m_textures[1] != nullptr && + (m_stageStates[1][D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == D3DTSS_TCI_CAMERASPACEPOSITION && + m_stageStates[1][D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2; + const bool tex_xform0 = layout.uv0 >= 0 && m_textures[0] != nullptr && + (m_stageStates[0][D3DTSS_TEXCOORDINDEX] & 0xFFFF0000u) == 0 && + m_stageStates[0][D3DTSS_TEXTURETRANSFORMFLAGS] == D3DTTFF_COUNT2; + GpuSkinSubrangeView skin_view; + const bool gpu_skinned = source_vertex != nullptr && + LookupGpuSkinSubrange(vertices, stride, lo, hi, &skin_view); + if (source_vertex) { - const uint8_t* v = vertices + (lo + i) * stride; - std::memcpy(&draw.positions[i * 3], v, 12); - if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4); - if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12); - if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8); - if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8); - if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4); + // The original buffers outlive individual draw calls. Keep their identity separate from + // Unlock revisions so a static surface can keep its Godot mesh across frames. + auto mix = [](std::uint64_t hash, std::uint64_t value) { + return (hash ^ value) * 1099511628211ull; + }; + std::uint64_t key = 14695981039346656037ull; + if (gpu_skinned) + { + key = mix(key, skin_view.source_mesh_key); + key = mix(key, source_index ? source_index->id : 0); + key = mix(key, source_first_index); + key = mix(key, lo - skin_view.mesh_base_vertex); + key = mix(key, hi - skin_view.mesh_base_vertex); + key = mix(key, primitives); + key = mix(key, type); + key = mix(key, stride); + key = mix(key, m_fvf); + draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull; + draw.geometry_revision = 1ull; + } + else + { + key = mix(key, source_vertex->id); + key = mix(key, source_index ? source_index->id : 0); + key = mix(key, source_first_index); + key = mix(key, lo); + key = mix(key, hi); + key = mix(key, primitives); + key = mix(key, type); + key = mix(key, stride); + key = mix(key, m_fvf); + if (texgen_cam_pos0) + key = mix(key, static_cast(reinterpret_cast(m_textures[0]))); + if (texgen_cam_pos1) + key = mix(key, static_cast(reinterpret_cast(m_textures[1]))); + draw.geometry_key = (key & 0x7FFFFFFFFFFFFFFFull) | 1ull; + std::uint64_t revision = mix(14695981039346656037ull, source_vertex->revision); + revision = mix(revision, source_index ? source_index->revision : 0); + if (tex_xform0) + { + const float* m = m_transforms[D3DTS_TEXTURE0]; + for (int mi : { 0, 1, 4, 5, 8, 9, 12, 13 }) + { + std::uint32_t bits = 0; + std::memcpy(&bits, &m[mi], sizeof(bits)); + revision = mix(revision, bits); + } + } + // Camera-space texture coordinates are baked into the captured geometry. + // The terrain splat and character-shadow matrices can change while the + // vertex/index buffers remain unchanged, so they are part of its revision. + for (int stage = 0; stage < 2; ++stage) + { + if ((stage == 0 && !texgen_cam_pos0) || (stage == 1 && !texgen_cam_pos1)) + continue; + float worldView[16], worldTexture[16]; + multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView); + multiply(worldView, m_transforms[stage == 0 ? D3DTS_TEXTURE0 : D3DTS_TEXTURE1], worldTexture); + for (int mi : { 0, 1, 4, 5, 8, 9, 12, 13 }) + { + std::uint32_t bits = 0; + std::memcpy(&bits, &worldTexture[mi], sizeof(bits)); + revision = mix(revision, bits); + } + } + draw.geometry_revision = revision & 0x7FFFFFFFFFFFFFFFull; + } + } + const std::uint64_t frame_id = UIRenderFrameId(); + auto cached = draw.geometry_key ? m_geometry_cache.find(draw.geometry_key) : m_geometry_cache.end(); + const bool reuse = cached != m_geometry_cache.end() && !cached->second.changing && + cached->second.revision == draw.geometry_revision; + if (reuse) + { + cached->second.last_used = frame_id; + const Render3DDraw& prior = cached->second.geometry; + draw.positions = prior.positions; + draw.rhw = prior.rhw; + draw.normals = prior.normals; + draw.uv0 = prior.uv0; + draw.uv1 = prior.uv1; + draw.diffuse = prior.diffuse; + draw.indices = prior.indices; + draw.bone_indices = prior.bone_indices; + draw.bone_weights = prior.bone_weights; + } + else + { + draw.positions.resize(count * 3); + if (layout.rhw) draw.rhw.resize(count); + if (layout.normal >= 0) draw.normals.resize(count * 3); + if (layout.uv0 >= 0) draw.uv0.resize(count * 2); + if (layout.uv1 >= 0) draw.uv1.resize(count * 2); + if (layout.diffuse >= 0) draw.diffuse.resize(count); + for (size_t i = 0; i < count; ++i) + { + const uint8_t* v = vertices + (lo + i) * stride; + std::memcpy(&draw.positions[i * 3], v, 12); + if (layout.rhw) std::memcpy(&draw.rhw[i], v + 12, 4); + if (layout.normal >= 0) std::memcpy(&draw.normals[i * 3], v + layout.normal, 12); + if (layout.uv0 >= 0) std::memcpy(&draw.uv0[i * 2], v + layout.uv0, 8); + if (layout.uv1 >= 0) std::memcpy(&draw.uv1[i * 2], v + layout.uv1, 8); + if (layout.diffuse >= 0) std::memcpy(&draw.diffuse[i], v + layout.diffuse, 4); + } + if (texgen_cam_pos0) + { + float worldView[16], worldTex0[16]; + multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView); + multiply(worldView, m_transforms[D3DTS_TEXTURE0], worldTex0); + draw.uv0.resize(count * 2); + const bool clamp0_u = m_stageStates[0][D3DTSS_ADDRESSU] == D3DTADDRESS_CLAMP; + const bool clamp0_v = m_stageStates[0][D3DTSS_ADDRESSV] == D3DTADDRESS_CLAMP; + for (size_t i = 0; i < count; ++i) + { + const float pos[4] = { draw.positions[i * 3 + 0], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f }; + float tc[4]; + transform_point(pos, worldTex0, tc); + draw.uv0[i * 2 + 0] = clamp0_u ? std::clamp(tc[0], 0.5f / 256.0f, 255.5f / 256.0f) : tc[0]; + draw.uv0[i * 2 + 1] = clamp0_v ? std::clamp(tc[1], 0.5f / 256.0f, 255.5f / 256.0f) : tc[1]; + } + } + else if (tex_xform0) + { + const float* m = m_transforms[D3DTS_TEXTURE0]; + for (size_t i = 0; i < count; ++i) + { + const float u = draw.uv0[i * 2 + 0]; + const float v = draw.uv0[i * 2 + 1]; + draw.uv0[i * 2 + 0] = u * m[0] + v * m[4] + m[8] + m[12]; + draw.uv0[i * 2 + 1] = u * m[1] + v * m[5] + m[9] + m[13]; + } + } + if (texgen_cam_pos1) + { + float worldView[16], worldTex1[16]; + multiply(m_transforms[256], m_transforms[D3DTS_VIEW], worldView); + multiply(worldView, m_transforms[D3DTS_TEXTURE1], worldTex1); + draw.uv1.resize(count * 2); + for (size_t i = 0; i < count; ++i) + { + const float pos[4] = { draw.positions[i * 3 + 0], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2], 1.0f }; + float tc[4]; + transform_point(pos, worldTex1, tc); + draw.uv1[i * 2 + 0] = tc[0]; + draw.uv1[i * 2 + 1] = tc[1]; + } + } + if (gpu_skinned) + { + const std::size_t rel_lo = std::size_t(lo - skin_view.mesh_base_vertex); + draw.bone_indices.resize(count * 4); + draw.bone_weights.resize(count * 4); + std::memcpy(draw.bone_indices.data(), skin_view.bone_indices + rel_lo * 4, count * 4 * sizeof(std::uint8_t)); + std::memcpy(draw.bone_weights.data(), skin_view.bone_weights + rel_lo * 4, count * 4 * sizeof(float)); + } + + // Strips and fans become lists; D3D flips the winding of every odd strip triangle. + switch (type) + { + case D3DPT_TRIANGLESTRIP: + for (UINT i = 0; i < primitives; ++i) + { + const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo; + if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c }); + else draw.indices.insert(draw.indices.end(), { a, b, c }); + } + break; + case D3DPT_TRIANGLEFAN: + for (UINT i = 0; i < primitives; ++i) + draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo }); + break; + case D3DPT_LINESTRIP: + for (UINT i = 0; i < primitives; ++i) + draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo }); + break; + default: + for (std::uint32_t index : indices) + draw.indices.push_back(index - lo); + break; + } + if (cached != m_geometry_cache.end()) + { + cached->second.last_used = frame_id; + if (!cached->second.changing && cached->second.revision != draw.geometry_revision) + { + cached->second.changing = true; + cached->second.geometry = Render3DDraw(); + } + } + else if (draw.geometry_key && cached == m_geometry_cache.end()) + m_geometry_cache.emplace(draw.geometry_key, GeometryCacheEntry{draw.geometry_revision, frame_id, false, draw}); + } + if (gpu_skinned && skin_view.bone_matrices && skin_view.bone_count > 0) + draw.bone_matrices.assign(skin_view.bone_matrices, skin_view.bone_matrices + std::size_t(skin_view.bone_count) * 16); + if (frame_id >= m_cache_pruned_frame + 120) + { + m_cache_pruned_frame = frame_id; + for (auto it = m_geometry_cache.begin(); it != m_geometry_cache.end();) + it = frame_id - it->second.last_used > 120 ? m_geometry_cache.erase(it) : std::next(it); } - // Strips and fans become lists; D3D flips the winding of every odd strip triangle. - switch (type) - { - case D3DPT_TRIANGLESTRIP: - for (UINT i = 0; i < primitives; ++i) - { - const std::uint32_t a = indices[i] - lo, b = indices[i + 1] - lo, c = indices[i + 2] - lo; - if (i & 1) draw.indices.insert(draw.indices.end(), { b, a, c }); - else draw.indices.insert(draw.indices.end(), { a, b, c }); - } - break; - case D3DPT_TRIANGLEFAN: - for (UINT i = 0; i < primitives; ++i) - draw.indices.insert(draw.indices.end(), { indices[0] - lo, indices[i + 1] - lo, indices[i + 2] - lo }); - break; - case D3DPT_LINESTRIP: - for (UINT i = 0; i < primitives; ++i) - draw.indices.insert(draw.indices.end(), { indices[i] - lo, indices[i + 1] - lo }); - break; - default: - for (std::uint32_t index : indices) - draw.indices.push_back(index - lo); - break; - } draw.alpha_blend = m_renderStates[D3DRS_ALPHABLENDENABLE]; draw.src_blend = m_renderStates[D3DRS_SRCBLEND]; @@ -682,7 +1023,17 @@ private: draw.lighting = m_renderStates[D3DRS_LIGHTING]; draw.texture_factor = m_renderStates[D3DRS_TEXTUREFACTOR]; draw.fog_enable = m_renderStates[D3DRS_FOGENABLE]; + draw.fog_color = m_renderStates[D3DRS_FOGCOLOR]; + draw.fog_vertex_mode = m_renderStates[D3DRS_FOGVERTEXMODE]; + draw.fog_table_mode = m_renderStates[D3DRS_FOGTABLEMODE]; + draw.fog_range_enable = m_renderStates[D3DRS_RANGEFOGENABLE]; + std::memcpy(&draw.fog_start, &m_renderStates[D3DRS_FOGSTART], sizeof(float)); + std::memcpy(&draw.fog_end, &m_renderStates[D3DRS_FOGEND], sizeof(float)); + std::memcpy(&draw.fog_density, &m_renderStates[D3DRS_FOGDENSITY], sizeof(float)); draw.ambient = m_renderStates[D3DRS_AMBIENT]; + draw.diffuse_material_source = m_renderStates[D3DRS_DIFFUSEMATERIALSOURCE]; + draw.ambient_material_source = m_renderStates[D3DRS_AMBIENTMATERIALSOURCE]; + draw.color_vertex = m_renderStates[D3DRS_COLORVERTEX]; for (int stage = 0; stage < 2; ++stage) { draw.color_op[stage] = m_stageStates[stage][D3DTSS_COLOROP]; @@ -691,11 +1042,16 @@ private: draw.alpha_op[stage] = m_stageStates[stage][D3DTSS_ALPHAOP]; draw.alpha_arg1[stage] = m_stageStates[stage][D3DTSS_ALPHAARG1]; draw.alpha_arg2[stage] = m_stageStates[stage][D3DTSS_ALPHAARG2]; + draw.address_u[stage] = m_stageStates[stage][D3DTSS_ADDRESSU]; + draw.address_v[stage] = m_stageStates[stage][D3DTSS_ADDRESSV]; + draw.min_filter[stage] = m_stageStates[stage][D3DTSS_MINFILTER]; + draw.mag_filter[stage] = m_stageStates[stage][D3DTSS_MAGFILTER]; + draw.mip_filter[stage] = m_stageStates[stage][D3DTSS_MIPFILTER]; } copy_color(draw.material_diffuse, m_material.Diffuse); copy_color(draw.material_ambient, m_material.Ambient); copy_color(draw.material_emissive, m_material.Emissive); - if (m_lightEnabled[0] && m_lights[0].Type == D3DLIGHT_DIRECTIONAL) + if (m_lightEnabled[0] && (m_lights[0].Type == D3DLIGHT_DIRECTIONAL || m_lights[0].Type == D3DLIGHT_SPOT)) { draw.light0 = true; draw.light0_direction[0] = m_lights[0].Direction.x; @@ -704,11 +1060,37 @@ private: copy_color(draw.light0_diffuse, m_lights[0].Diffuse); copy_color(draw.light0_ambient, m_lights[0].Ambient); } - Render3DAdd(std::move(draw)); + for (int i = 0; i < 2; ++i) + { + if (!m_lightEnabled[i]) continue; + const D3DLIGHT8& source = m_lights[i]; + auto& light = draw.lights[i]; + light.type = source.Type; + light.position[0] = source.Position.x; + light.position[1] = source.Position.y; + light.position[2] = source.Position.z; + light.direction[0] = source.Direction.x; + light.direction[1] = source.Direction.y; + light.direction[2] = source.Direction.z; + copy_color(light.diffuse, source.Diffuse); + copy_color(light.ambient, source.Ambient); + light.attenuation[0] = source.Attenuation0; + light.attenuation[1] = source.Attenuation1; + light.attenuation[2] = source.Attenuation2; + light.range = source.Range; + light.theta = source.Theta; + light.phi = source.Phi; + light.falloff = source.Falloff; + } + if (m_renderTarget == m_backBuffer) + Render3DAdd(std::move(draw)); + else + rasterize_shadow(draw); } ULONG m_refs = 1; int m_width, m_height; + D3DVIEWPORT8 m_viewport = {}; float m_transforms[kTransforms][16]; DWORD m_renderStates[kRenderStates] = {}; DWORD m_stageStates[kStages][kStageStates] = {}; @@ -725,11 +1107,103 @@ private: UINT m_streamStride = 0; MtCpuIndexBuffer* m_indices = nullptr; UINT m_baseVertex = 0; + struct GeometryCacheEntry { + std::uint64_t revision; + std::uint64_t last_used; + bool changing; + Render3DDraw geometry; + }; + std::unordered_map m_geometry_cache; + std::uint64_t m_cache_pruned_frame = 0; + std::vector m_offscreenDepth; }; } +std::string MtCpuTextureNameFromHandle(const IDirect3DBaseTexture8* handle) +{ + if (!handle) return {}; + std::lock_guard lock(g_cpu_tex_mutex); + for (const auto& entry : g_cpu_textures) + { + if (entry.second == handle && !entry.second->levels.empty()) + return "mem:cpu_" + std::to_string(entry.first) + "@" + std::to_string(entry.second->revision); + } + return {}; +} + +bool MtCpuMemoryTexture(const std::string& name, UIMemoryTexture* out) +{ + if (name.compare(0, 8, "mem:cpu_") != 0 || !out) return false; + const unsigned id = unsigned(std::strtoul(name.c_str() + 8, nullptr, 10)); + std::lock_guard lock(g_cpu_tex_mutex); + auto it = g_cpu_textures.find(id); + if (it == g_cpu_textures.end() || it->second->levels.empty()) return false; + const auto& lv0 = it->second->levels[0]; + const UINT w = lv0.desc.Width; + const UINT h = lv0.desc.Height; + if (!w || !h) return false; + out->width = static_cast(w); + out->height = static_cast(h); + out->revision = it->second->revision; + out->argb.resize(std::size_t(w) * std::size_t(h)); + const uint8_t* bytes = lv0.bytes.data(); + const int bpp = format_bytes(lv0.desc.Format); + for (std::size_t i = 0; i < out->argb.size(); ++i) + { + if (bpp == 4) + { + std::uint32_t px = 0; + std::memcpy(&px, bytes + i * 4, 4); + // CTerrain::PutImage32 packs alpha into bits 31..24 with RGB == 0. + // Set RGB to white so stage-1 modulation preserves the base texture's RGB. + if ((px & 0x00FFFFFFu) == 0) + px |= 0x00FFFFFFu; + out->argb[i] = px; + } + else if (lv0.desc.Format == D3DFMT_R5G6B5) + { + std::uint16_t word = 0; + std::memcpy(&word, bytes + i * 2, 2); + const std::uint32_t r = ((word >> 11) & 31u) * 255u / 31u; + const std::uint32_t g = ((word >> 5) & 63u) * 255u / 63u; + const std::uint32_t b = (word & 31u) * 255u / 31u; + out->argb[i] = 0xFF000000u | (r << 16) | (g << 8) | b; + } + else if (bpp == 2) + { + std::uint16_t word = 0; + std::memcpy(&word, bytes + i * 2, 2); + // CTerrain::PutImage16 writes `src[x] << 8`, storing the full 8-bit alpha in the high byte. + const std::uint32_t a = (word >> 8) & 0xFFu; + out->argb[i] = (a << 24) | 0x00FFFFFFu; + } + else + { + const std::uint32_t a = bytes[i]; + out->argb[i] = (a << 24) | 0x00FFFFFFu; + } + } + return true; +} + IDirect3DDevice8* MtCreateRecordingDevice(int width, int height) { return new RecordingDevice(width, height); } +void SetNativeTerrainRenderEnabled(bool enabled) +{ + g_native_terrain_override = enabled ? 1 : 0; +} + +bool IsNativeTerrainRenderEnabled() +{ + if (g_native_terrain_override >= 0) + return g_native_terrain_override != 0; + static const bool env_on = [] { + const char* v = std::getenv("MT_NATIVE_TERRAIN"); + return v && (*v == '1' || *v == 't' || *v == 'T' || *v == 'y' || *v == 'Y'); + }(); + return env_on; +} + void Render3DBeginFrame() { std::lock_guard lock(g_draws_mutex); diff --git a/extension/src/platform/EterLib/RenderCommands3D.h b/extension/src/platform/EterLib/RenderCommands3D.h index 8b740e44..859728dc 100644 --- a/extension/src/platform/EterLib/RenderCommands3D.h +++ b/extension/src/platform/EterLib/RenderCommands3D.h @@ -10,9 +10,14 @@ // // Matrices are D3D8's row-vector layout (translation in elements 12..14), exactly as 40250 set them. struct Render3DDraw { + // Stable source-buffer identity and content version. Zero key means an immediate-mode draw + // without a reusable D3D buffer. Geometry may be reused only while both values match. + std::uint64_t geometry_key = 0; + std::uint64_t geometry_revision = 0; float world[16]; float view[16]; float proj[16]; + float viewport[4] = {}; // x, y, width, height // Stage 0/1 textures, named like UIRenderTextureName: the pack path of a file texture or // "mem:@"; empty when the stage has no texture. @@ -30,6 +35,9 @@ struct Render3DDraw { std::vector uv1; // u, v std::vector diffuse; // 0xAARRGGBB std::vector indices; // triangle list (strips and fans are expanded), or line list + std::vector bone_indices; // 4 uint8 per vertex (mesh-local bone palette index) + std::vector bone_weights; // 4 floats per vertex + std::vector bone_matrices; // 16 floats per bone in the mesh's palette (D3D row-vector layout) bool lines = false; // D3DRS_* / D3DTSS_* values in effect for the draw. @@ -37,8 +45,12 @@ struct Render3DDraw { std::uint32_t alpha_test = 0, alpha_ref = 0, alpha_func = 0; std::uint32_t cull_mode = 0, z_enable = 0, z_write = 0, z_func = 0; std::uint32_t lighting = 0, texture_factor = 0, fog_enable = 0; + std::uint32_t fog_color = 0, fog_vertex_mode = 0, fog_table_mode = 0, fog_range_enable = 0; + float fog_start = 0, fog_end = 0, fog_density = 0; std::uint32_t color_op[2] = {}, color_arg1[2] = {}, color_arg2[2] = {}; std::uint32_t alpha_op[2] = {}, alpha_arg1[2] = {}, alpha_arg2[2] = {}; + std::uint32_t address_u[2] = {}, address_v[2] = {}; + std::uint32_t min_filter[2] = {}, mag_filter[2] = {}, mip_filter[2] = {}; // D3DMATERIAL8 diffuse/ambient/emissive (r, g, b, a) and light 0 when enabled. float material_diffuse[4] = {1, 1, 1, 1}; float material_ambient[4] = {}; @@ -48,8 +60,40 @@ struct Render3DDraw { float light0_diffuse[4] = {}; float light0_ambient[4] = {}; std::uint32_t ambient = 0; // D3DRS_AMBIENT + struct Light { + std::uint32_t type = 0; // zero when disabled; D3DLIGHT_POINT/SPOT/DIRECTIONAL otherwise + float position[3] = {}; + float direction[3] = {}; + float diffuse[4] = {}; + float ambient[4] = {}; + float attenuation[3] = {}; + float range = 0; + float theta = 0, phi = 0, falloff = 0; + } lights[2]; + std::uint32_t diffuse_material_source = 0; + std::uint32_t ambient_material_source = 0; + std::uint32_t color_vertex = 0; }; +struct GpuSkinSubrangeView { + std::uint64_t source_mesh_key = 0; + std::uint32_t mesh_base_vertex = 0; + std::uint32_t mesh_vertex_count = 0; + const std::uint8_t* bone_indices = nullptr; // mesh_vertex_count * 4 + const float* bone_weights = nullptr; // mesh_vertex_count * 4 + const float* bone_matrices = nullptr; // bone_count * 16 + std::uint32_t bone_count = 0; +}; + +void SetNativeTerrainRenderEnabled(bool enabled); +bool IsNativeTerrainRenderEnabled(); + +void SetGpuSkinningEnabled(bool enabled); +bool IsGpuSkinningEnabled(); +bool LookupGpuSkinSubrange(const void* vertex_buffer_base, std::uint32_t stride, + std::uint32_t lo_vertex, std::uint32_t hi_vertex, + GpuSkinSubrangeView* out_view); + void Render3DBeginFrame(); void Render3DAdd(Render3DDraw draw); const std::vector& Render3DDraws(); diff --git a/extension/src/platform/EterLib/SkyBox.cpp b/extension/src/platform/EterLib/SkyBox.cpp index c28b69d6..45a78d62 100644 --- a/extension/src/platform/EterLib/SkyBox.cpp +++ b/extension/src/platform/EterLib/SkyBox.cpp @@ -1,199 +1,787 @@ -// Platform skeleton for EterLib/SkyBox.h (40250 EterLib/SkyBox.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. +// Platform implementation of EterLib/SkyBox.cpp (40250 EterLib/SkyBox.cpp). #include "EterLib/StdAfx.h" #include "EterLib/SkyBox.h" - -#include "../PlatformStub.h" +#include "EterLib/Camera.h" +#include "EterLib/StateManager.h" +#include "EterLib/ResourceManager.h" +#include "EterBase/Timer.h" +#include "RenderCommands3D.h" CSkyObjectQuad::CSkyObjectQuad() { - MT_PLATFORM_STUB(); + m_Indices[0] = 0; + m_Indices[1] = 2; + m_Indices[2] = 1; + m_Indices[3] = 3; + + for (unsigned char uci = 0; uci < 4; ++uci) + { + memset(&m_Vertex[uci], 0, sizeof(TPDTVertex)); + } } CSkyObjectQuad::~CSkyObjectQuad() { - MT_PLATFORM_STUB(); } -auto CSkyObjectQuad::Clear(const unsigned char &, const float &, const float &, const float &, const float &) -> void +void CSkyObjectQuad::Clear(const unsigned char & c_rucNumVertex, + const float & c_rfRed, + const float & c_rfGreen, + const float & c_rfBlue, + const float & c_rfAlpha) { - MT_PLATFORM_STUB(); + if (c_rucNumVertex > 3) + return; + m_Helper[c_rucNumVertex].Clear(c_rfRed, c_rfGreen, c_rfBlue, c_rfAlpha); } -auto CSkyObjectQuad::SetSrcColor(const unsigned char &, const float &, const float &, const float &, const float &) -> void +void CSkyObjectQuad::SetSrcColor(const unsigned char & c_rucNumVertex, + const float & c_rfRed, + const float & c_rfGreen, + const float & c_rfBlue, + const float & c_rfAlpha) { - MT_PLATFORM_STUB(); + if (c_rucNumVertex > 3) + return; + m_Helper[c_rucNumVertex].SetSrcColor(c_rfRed, c_rfGreen, c_rfBlue, c_rfAlpha); } -auto CSkyObjectQuad::SetTransition(const unsigned char &, const float &, const float &, const float &, const float &, DWORD) -> void +void CSkyObjectQuad::SetTransition(const unsigned char & c_rucNumVertex, + const float & c_rfRed, + const float & c_rfGreen, + const float & c_rfBlue, + const float & c_rfAlpha, + DWORD dwDuration) { - MT_PLATFORM_STUB(); + if (c_rucNumVertex > 3) + return; + m_Helper[c_rucNumVertex].SetTransition(c_rfRed, c_rfGreen, c_rfBlue, c_rfAlpha, dwDuration); } -auto CSkyObjectQuad::SetVertex(const unsigned char &, const TPDTVertex &) -> void +void CSkyObjectQuad::SetVertex(const unsigned char & c_rucNumVertex, const TPDTVertex & c_rPDTVertex) { - MT_PLATFORM_STUB(); + if (c_rucNumVertex > 3) + return; + memcpy(&m_Vertex[m_Indices[c_rucNumVertex]], &c_rPDTVertex, sizeof(TPDTVertex)); } -auto CSkyObjectQuad::StartTransition() -> void +void CSkyObjectQuad::StartTransition() { - MT_PLATFORM_STUB(); + for (unsigned char uci = 0; uci < 4; ++uci) + { + m_Helper[uci].StartTransition(); + } } -auto CSkyObjectQuad::Update() -> bool +bool CSkyObjectQuad::Update() { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + bool bResult = false; + for (unsigned char uci = 0; uci < 4; ++uci) + { + bResult = m_Helper[uci].Update() || bResult; + m_Vertex[m_Indices[uci]].diffuse = m_Helper[uci].GetCurColor(); + } + return bResult; } -auto CSkyObjectQuad::Render() -> void +void CSkyObjectQuad::Render() { - MT_PLATFORM_STUB(); + if (CGraphicBase::SetPDTStream(m_Vertex, 4)) + STATEMANAGER.DrawPrimitive(D3DPT_TRIANGLESTRIP, 0, 2); } -CSkyObject::CSkyObject() +CSkyObject::CSkyObject() : + m_v3Position(0.0f, 0.0f, 0.0f), + m_fScaleX(1.0f), + m_fScaleY(1.0f), + m_fScaleZ(1.0f) { - MT_PLATFORM_STUB(); + D3DXMatrixIdentity(&m_matWorld); + D3DXMatrixIdentity(&m_matTranslation); + D3DXMatrixIdentity(&m_matWorldCloud); + D3DXMatrixIdentity(&m_matTranslationCloud); + D3DXMatrixIdentity(&m_matTextureCloud); + + m_dwlastTime = CTimer::Instance().GetCurrentMillisecond(); + + m_fCloudPositionU = 0.0f; + m_fCloudPositionV = 0.0f; + m_fCloudScaleX = 1.0f; + m_fCloudScaleY = 1.0f; + m_fCloudHeight = 0.0f; + m_fCloudTextureScaleX = 1.0f; + m_fCloudTextureScaleY = 1.0f; + m_fCloudScrollSpeedU = 0.0f; + m_fCloudScrollSpeedV = 0.0f; + m_ucRenderMode = SKY_RENDER_MODE_DEFAULT; + + m_bTransitionStarted = false; + m_bSkyMatrixUpdated = false; } CSkyObject::~CSkyObject() { - MT_PLATFORM_STUB(); } -auto CSkyObject::StartTransition() -> void +void CSkyObject::Destroy() { - MT_PLATFORM_STUB(); } -auto CSkyObject::GenerateTexture(const char *) -> CGraphicImageInstance * +void CSkyObject::Update() { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + CCamera* pCamera = CCameraManager::Instance().GetCurrentCamera(); + if (!pCamera) + return; + D3DXVECTOR3 v3Eye = pCamera->GetEye(); + + if (m_v3Position == v3Eye) + if (m_bSkyMatrixUpdated == false) + return; + + m_v3Position = v3Eye; + + m_matWorld._41 = m_v3Position.x; + m_matWorld._42 = m_v3Position.y; + m_matWorld._43 = m_v3Position.z; + + m_matWorldCloud._41 = m_v3Position.x; + m_matWorldCloud._42 = m_v3Position.y; + m_matWorldCloud._43 = m_v3Position.z + m_fCloudHeight; + + if (m_bSkyMatrixUpdated) + m_bSkyMatrixUpdated = false; } -auto CSkyObject::DeleteTexture(CGraphicImageInstance *) -> void +void CSkyObject::Render() { - MT_PLATFORM_STUB(); } -auto CSkyObject::CSkyBox::StartTransition() -> void +void CSkyObject::StartTransition() { - MT_PLATFORM_STUB(); } -auto CSkyObject::CSkyBox::Update() -> bool +CGraphicImageInstance * CSkyObject::GenerateTexture(const char * szfilename) { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + if (!szfilename || !*szfilename) + return NULL; + + CResource * pResource = CResourceManager::Instance().GetResourcePointer(szfilename); + if (!pResource || !pResource->IsType(CGraphicImage::Type())) + return NULL; + + CGraphicImageInstance * pImageInstance = CGraphicImageInstance::New(); + pImageInstance->SetImagePointer(static_cast(pResource)); + return pImageInstance; } -auto CSkyObject::CSkyBox::Render() -> void +void CSkyObject::DeleteTexture(CGraphicImageInstance * pImageInstance) { - MT_PLATFORM_STUB(); + if (pImageInstance) + CGraphicImageInstance::Delete(pImageInstance); +} + +void CSkyObject::TSkyObjectFace::StartTransition() +{ + for (unsigned char uci = 0; uci < m_SkyObjectQuadVector.size(); ++uci) + { + m_SkyObjectQuadVector[uci].StartTransition(); + } +} + +bool CSkyObject::TSkyObjectFace::Update() +{ + bool bResult = false; + for (DWORD dwi = 0; dwi < m_SkyObjectQuadVector.size(); ++dwi) + bResult = m_SkyObjectQuadVector[dwi].Update() || bResult; + return bResult; +} + +void CSkyObject::TSkyObjectFace::Render() +{ + for (unsigned char uci = 0; uci < m_SkyObjectQuadVector.size(); ++uci) + { + m_SkyObjectQuadVector[uci].Render(); + } } CSkyBox::CSkyBox() { - MT_PLATFORM_STUB(); + m_ucVirticalGradientLevelUpper = 0; + m_ucVirticalGradientLevelLower = 0; } CSkyBox::~CSkyBox() { - MT_PLATFORM_STUB(); + Destroy(); } -auto CSkyBox::Update() -> void +void CSkyBox::Destroy() { - MT_PLATFORM_STUB(); + Unload(); } -auto CSkyBox::Render() -> void +void CSkyBox::Unload() { - MT_PLATFORM_STUB(); + TGraphicImageInstanceMap::iterator itor = m_GraphicImageInstanceMap.begin(); + + while (itor != m_GraphicImageInstanceMap.end()) + { + DeleteTexture(itor->second); + ++itor; + } + + m_GraphicImageInstanceMap.clear(); } -auto CSkyBox::RenderCloud() -> void +void CSkyBox::SetSkyBoxScale(const D3DXVECTOR3 & c_rv3Scale) { - MT_PLATFORM_STUB(); + m_fScaleX = c_rv3Scale.x; + m_fScaleY = c_rv3Scale.y; + m_fScaleZ = c_rv3Scale.z; + + m_bSkyMatrixUpdated = true; + D3DXMatrixScaling(&m_matWorld, m_fScaleX, m_fScaleY, m_fScaleZ); } -auto CSkyBox::Destroy() -> void +void CSkyBox::SetGradientLevel(BYTE byUpper, BYTE byLower) { - MT_PLATFORM_STUB(); + m_ucVirticalGradientLevelUpper = byUpper; + m_ucVirticalGradientLevelLower = byLower; } -auto CSkyBox::Unload() -> void +void CSkyBox::SetFaceTexture(const char* c_szFileName, int iFaceIndex) { - MT_PLATFORM_STUB(); + if (iFaceIndex < 0 || iFaceIndex > 5 || !c_szFileName || !*c_szFileName) + return; + + TGraphicImageInstanceMap::iterator itor = m_GraphicImageInstanceMap.find(c_szFileName); + if (m_GraphicImageInstanceMap.end() != itor) + return; + + m_Faces[iFaceIndex].m_strFaceTextureFileName = c_szFileName; + + CGraphicImageInstance * pGraphicImageInstance = GenerateTexture(c_szFileName); + m_GraphicImageInstanceMap.insert(TGraphicImageInstanceMap::value_type(c_szFileName, pGraphicImageInstance)); } -auto CSkyBox::SetSkyBoxScale(const D3DXVECTOR3 &) -> void +void CSkyBox::SetCloudTexture(const char * c_szFileName) { - MT_PLATFORM_STUB(); + if (!c_szFileName || !*c_szFileName) + return; + + TGraphicImageInstanceMap::iterator itor = m_GraphicImageInstanceMap.find(c_szFileName); + if (m_GraphicImageInstanceMap.end() != itor) + return; + + m_FaceCloud.m_strfacename = c_szFileName; + CGraphicImageInstance * pGraphicImageInstance = GenerateTexture(c_szFileName); + m_GraphicImageInstanceMap.insert(TGraphicImageInstanceMap::value_type(m_FaceCloud.m_strfacename, pGraphicImageInstance)); } -auto CSkyBox::SetGradientLevel(BYTE, BYTE) -> void +void CSkyBox::SetCloudScale(const D3DXVECTOR2 & c_rv2CloudScale) { - MT_PLATFORM_STUB(); + m_fCloudScaleX = c_rv2CloudScale.x; + m_fCloudScaleY = c_rv2CloudScale.y; + + D3DXMatrixScaling(&m_matWorldCloud, m_fCloudScaleX, m_fCloudScaleY, 1.0f); } -auto CSkyBox::SetFaceTexture(const char *, int) -> void +void CSkyBox::SetCloudHeight(float fHeight) { - MT_PLATFORM_STUB(); + m_fCloudHeight = fHeight; } -auto CSkyBox::SetCloudTexture(const char *) -> void +void CSkyBox::SetCloudTextureScale(const D3DXVECTOR2 & c_rv2CloudTextureScale) { - MT_PLATFORM_STUB(); + m_fCloudTextureScaleX = c_rv2CloudTextureScale.x; + m_fCloudTextureScaleY = c_rv2CloudTextureScale.y; + + m_matTextureCloud._11 = m_fCloudTextureScaleX; + m_matTextureCloud._22 = m_fCloudTextureScaleY; } -auto CSkyBox::SetCloudScale(const D3DXVECTOR2 &) -> void +void CSkyBox::SetCloudScrollSpeed(const D3DXVECTOR2 & c_rv2CloudScrollSpeed) { - MT_PLATFORM_STUB(); + m_fCloudScrollSpeedU = c_rv2CloudScrollSpeed.x; + m_fCloudScrollSpeedV = c_rv2CloudScrollSpeed.y; } -auto CSkyBox::SetCloudHeight(float) -> void +void CSkyBox::SetSkyObjectQuadVertical(TSkyObjectQuadVector * pSkyObjectQuadVector, const D3DXVECTOR2 * c_pv2QuadPoints) { - MT_PLATFORM_STUB(); + TPDTVertex aPDTVertex; + + DWORD dwIndex = 0; + + pSkyObjectQuadVector->clear(); + pSkyObjectQuadVector->resize(m_ucVirticalGradientLevelUpper + m_ucVirticalGradientLevelLower); + + unsigned char ucY; + for (ucY = 0; ucY < m_ucVirticalGradientLevelUpper; ++ucY) + { + CSkyObjectQuad & rSkyObjectQuad = pSkyObjectQuadVector->at(dwIndex++); + + aPDTVertex.position.x = c_pv2QuadPoints[0].x; + aPDTVertex.position.y = c_pv2QuadPoints[0].y; + aPDTVertex.position.z = 1.0f - (float)(ucY + 1) / (float)(m_ucVirticalGradientLevelUpper); + aPDTVertex.texCoord.x = 0.0f; + aPDTVertex.texCoord.y = (float)(ucY + 1) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(0, aPDTVertex); + aPDTVertex.position.x = c_pv2QuadPoints[0].x; + aPDTVertex.position.y = c_pv2QuadPoints[0].y; + aPDTVertex.position.z = 1.0f - (float)(ucY) / (float)(m_ucVirticalGradientLevelUpper); + aPDTVertex.texCoord.x = 0.0f; + aPDTVertex.texCoord.y = (float)(ucY) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(1, aPDTVertex); + aPDTVertex.position.x = c_pv2QuadPoints[1].x; + aPDTVertex.position.y = c_pv2QuadPoints[1].y; + aPDTVertex.position.z = 1.0f - (float)(ucY + 1) / (float)(m_ucVirticalGradientLevelUpper); + aPDTVertex.texCoord.x = 1.0f; + aPDTVertex.texCoord.y = (float)(ucY + 1) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(2, aPDTVertex); + aPDTVertex.position.x = c_pv2QuadPoints[1].x; + aPDTVertex.position.y = c_pv2QuadPoints[1].y; + aPDTVertex.position.z = 1.0f - (float)(ucY) / (float)(m_ucVirticalGradientLevelUpper); + aPDTVertex.texCoord.x = 1.0f; + aPDTVertex.texCoord.y = (float)(ucY) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(3, aPDTVertex); + } + for (ucY = 0; ucY < m_ucVirticalGradientLevelLower; ++ucY) + { + CSkyObjectQuad & rSkyObjectQuad = pSkyObjectQuadVector->at(dwIndex++); + + aPDTVertex.position.x = c_pv2QuadPoints[0].x; + aPDTVertex.position.y = c_pv2QuadPoints[0].y; + aPDTVertex.position.z = -(float)(ucY + 1) / (float)(m_ucVirticalGradientLevelLower); + aPDTVertex.texCoord.x = 0.0f; + aPDTVertex.texCoord.y = 0.5f + (float)(ucY + 1) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(0, aPDTVertex); + aPDTVertex.position.x = c_pv2QuadPoints[0].x; + aPDTVertex.position.y = c_pv2QuadPoints[0].y; + aPDTVertex.position.z = -(float)(ucY) / (float)(m_ucVirticalGradientLevelLower); + aPDTVertex.texCoord.x = 0.0f; + aPDTVertex.texCoord.y = 0.5f + (float)(ucY) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(1, aPDTVertex); + aPDTVertex.position.x = c_pv2QuadPoints[1].x; + aPDTVertex.position.y = c_pv2QuadPoints[1].y; + aPDTVertex.position.z = -(float)(ucY + 1) / (float)(m_ucVirticalGradientLevelLower); + aPDTVertex.texCoord.x = 1.0f; + aPDTVertex.texCoord.y = 0.5f + (float)(ucY + 1) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(2, aPDTVertex); + aPDTVertex.position.x = c_pv2QuadPoints[1].x; + aPDTVertex.position.y = c_pv2QuadPoints[1].y; + aPDTVertex.position.z = -(float)(ucY) / (float)(m_ucVirticalGradientLevelLower); + aPDTVertex.texCoord.x = 1.0f; + aPDTVertex.texCoord.y = 0.5f + (float)(ucY) / (float)(m_ucVirticalGradientLevelUpper) * 0.5f; + rSkyObjectQuad.SetVertex(3, aPDTVertex); + } } -auto CSkyBox::SetCloudTextureScale(const D3DXVECTOR2 &) -> void +void CSkyBox::SetSkyObjectQuadHorizon(TSkyObjectQuadVector * pSkyObjectQuadVector, const D3DXVECTOR3 * c_pv3QuadPoints) { - MT_PLATFORM_STUB(); + pSkyObjectQuadVector->clear(); + pSkyObjectQuadVector->resize(1); + CSkyObjectQuad & rSkyObjectQuad = pSkyObjectQuadVector->at(0); + + TPDTVertex aPDTVertex{}; + aPDTVertex.position = c_pv3QuadPoints[0]; + aPDTVertex.texCoord.x = 0.0f; + aPDTVertex.texCoord.y = 1.0f; + rSkyObjectQuad.SetVertex(0, aPDTVertex); + + aPDTVertex.position = c_pv3QuadPoints[1]; + aPDTVertex.texCoord.x = 0.0f; + aPDTVertex.texCoord.y = 0.0f; + rSkyObjectQuad.SetVertex(1, aPDTVertex); + + aPDTVertex.position = c_pv3QuadPoints[2]; + aPDTVertex.texCoord.x = 1.0f; + aPDTVertex.texCoord.y = 1.0f; + rSkyObjectQuad.SetVertex(2, aPDTVertex); + + aPDTVertex.position = c_pv3QuadPoints[3]; + aPDTVertex.texCoord.x = 1.0f; + aPDTVertex.texCoord.y = 0.0f; + rSkyObjectQuad.SetVertex(3, aPDTVertex); } -auto CSkyBox::SetCloudScrollSpeed(const D3DXVECTOR2 &) -> void +void CSkyBox::Refresh() { - MT_PLATFORM_STUB(); + D3DXVECTOR3 v3QuadPoints[4]; + + if (m_ucRenderMode == CSkyObject::SKY_RENDER_MODE_DEFAULT || m_ucRenderMode == CSkyObject::SKY_RENDER_MODE_DIFFUSE) + { + if (m_ucVirticalGradientLevelUpper + m_ucVirticalGradientLevelLower <= 0) + return; + + D3DXVECTOR2 v2QuadPoints[2]; + + v2QuadPoints[0] = D3DXVECTOR2(1.0f, -1.0f); + v2QuadPoints[1] = D3DXVECTOR2(-1.0f, -1.0f); + SetSkyObjectQuadVertical(&m_Faces[0].m_SkyObjectQuadVector, v2QuadPoints); + m_Faces[0].m_strfacename = "front"; + + v2QuadPoints[0] = D3DXVECTOR2(-1.0f, 1.0f); + v2QuadPoints[1] = D3DXVECTOR2(1.0f, 1.0f); + SetSkyObjectQuadVertical(&m_Faces[1].m_SkyObjectQuadVector, v2QuadPoints); + m_Faces[1].m_strfacename = "back"; + + v2QuadPoints[0] = D3DXVECTOR2(-1.0f, -1.0f); + v2QuadPoints[1] = D3DXVECTOR2(-1.0f, 1.0f); + SetSkyObjectQuadVertical(&m_Faces[2].m_SkyObjectQuadVector, v2QuadPoints); + m_Faces[2].m_strfacename = "left"; + + v2QuadPoints[0] = D3DXVECTOR2(1.0f, 1.0f); + v2QuadPoints[1] = D3DXVECTOR2(1.0f, -1.0f); + SetSkyObjectQuadVertical(&m_Faces[3].m_SkyObjectQuadVector, v2QuadPoints); + m_Faces[3].m_strfacename = "right"; + + v3QuadPoints[0] = D3DXVECTOR3(1.0f, 1.0f, 1.0f); + v3QuadPoints[1] = D3DXVECTOR3(-1.0f, 1.0f, 1.0f); + v3QuadPoints[2] = D3DXVECTOR3(1.0f, -1.0f, 1.0f); + v3QuadPoints[3] = D3DXVECTOR3(-1.0f, -1.0f, 1.0f); + SetSkyObjectQuadHorizon(&m_Faces[4].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[4].m_strfacename = "top"; + + v3QuadPoints[0] = D3DXVECTOR3(-1.0f, 1.0f, -1.0f); + v3QuadPoints[1] = D3DXVECTOR3(1.0f, 1.0f, -1.0f); + v3QuadPoints[2] = D3DXVECTOR3(-1.0f, -1.0f, -1.0f); + v3QuadPoints[3] = D3DXVECTOR3(1.0f, -1.0f, -1.0f); + SetSkyObjectQuadHorizon(&m_Faces[5].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[5].m_strfacename = "bottom"; + } + else if (m_ucRenderMode == CSkyObject::SKY_RENDER_MODE_TEXTURE) + { + v3QuadPoints[0] = D3DXVECTOR3(1.0f, -1.0f, -1.0f); + v3QuadPoints[1] = D3DXVECTOR3(1.0f, -1.0f, 1.0f); + v3QuadPoints[2] = D3DXVECTOR3(-1.0f, -1.0f, -1.0f); + v3QuadPoints[3] = D3DXVECTOR3(-1.0f, -1.0f, 1.0f); + SetSkyObjectQuadHorizon(&m_Faces[0].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[0].m_strfacename = "front"; + + v3QuadPoints[0] = D3DXVECTOR3(-1.0f, 1.0f, -1.0f); + v3QuadPoints[1] = D3DXVECTOR3(-1.0f, 1.0f, 1.0f); + v3QuadPoints[2] = D3DXVECTOR3(1.0f, 1.0f, -1.0f); + v3QuadPoints[3] = D3DXVECTOR3(1.0f, 1.0f, 1.0f); + SetSkyObjectQuadHorizon(&m_Faces[1].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[1].m_strfacename = "back"; + + v3QuadPoints[0] = D3DXVECTOR3(1.0f, 1.0f, -1.0f); + v3QuadPoints[1] = D3DXVECTOR3(1.0f, 1.0f, 1.0f); + v3QuadPoints[2] = D3DXVECTOR3(1.0f, -1.0f, -1.0f); + v3QuadPoints[3] = D3DXVECTOR3(1.0f, -1.0f, 1.0f); + SetSkyObjectQuadHorizon(&m_Faces[2].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[2].m_strfacename = "left"; + + v3QuadPoints[0] = D3DXVECTOR3(-1.0f, -1.0f, -1.0f); + v3QuadPoints[1] = D3DXVECTOR3(-1.0f, -1.0f, 1.0f); + v3QuadPoints[2] = D3DXVECTOR3(-1.0f, 1.0f, -1.0f); + v3QuadPoints[3] = D3DXVECTOR3(-1.0f, 1.0f, 1.0f); + SetSkyObjectQuadHorizon(&m_Faces[3].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[3].m_strfacename = "right"; + + v3QuadPoints[0] = D3DXVECTOR3(1.0f, -1.0f, 1.0f); + v3QuadPoints[1] = D3DXVECTOR3(1.0f, 1.0f, 1.0f); + v3QuadPoints[2] = D3DXVECTOR3(-1.0f, -1.0f, 1.0f); + v3QuadPoints[3] = D3DXVECTOR3(-1.0f, 1.0f, 1.0f); + SetSkyObjectQuadHorizon(&m_Faces[4].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[4].m_strfacename = "top"; + + v3QuadPoints[0] = D3DXVECTOR3(1.0f, -1.0f, -1.0f); + v3QuadPoints[1] = D3DXVECTOR3(1.0f, 1.0f, -1.0f); + v3QuadPoints[2] = D3DXVECTOR3(-1.0f, -1.0f, -1.0f); + v3QuadPoints[3] = D3DXVECTOR3(-1.0f, 1.0f, -1.0f); + SetSkyObjectQuadHorizon(&m_Faces[5].m_SkyObjectQuadVector, v3QuadPoints); + m_Faces[5].m_strfacename = "bottom"; + } + + v3QuadPoints[0] = D3DXVECTOR3(1.0f, 1.0f, 0.0f); + v3QuadPoints[1] = D3DXVECTOR3(-1.0f, 1.0f, 0.0f); + v3QuadPoints[2] = D3DXVECTOR3(1.0f, -1.0f, 0.0f); + v3QuadPoints[3] = D3DXVECTOR3(-1.0f, -1.0f, 0.0f); + SetSkyObjectQuadHorizon(&m_FaceCloud.m_SkyObjectQuadVector, v3QuadPoints); } -auto CSkyBox::SetCloudColor(const TGradientColor &, const TGradientColor &, const DWORD &) -> void +void CSkyBox::SetCloudColor(const TGradientColor & c_rColor, const TGradientColor & c_rNextColor, const DWORD & dwTransitionTime) { - MT_PLATFORM_STUB(); + TSkyObjectFace & aFaceCloud = m_FaceCloud; + for (DWORD dwk = 0; dwk < aFaceCloud.m_SkyObjectQuadVector.size(); ++dwk) + { + CSkyObjectQuad & aSkyObjectQuad = aFaceCloud.m_SkyObjectQuadVector[dwk]; + for (unsigned char v = 0; v < 4; ++v) + { + aSkyObjectQuad.SetSrcColor(v, + c_rColor.m_FirstColor.r, + c_rColor.m_FirstColor.g, + c_rColor.m_FirstColor.b, + c_rColor.m_FirstColor.a); + aSkyObjectQuad.SetTransition(v, + c_rNextColor.m_FirstColor.r, + c_rNextColor.m_FirstColor.g, + c_rNextColor.m_FirstColor.b, + c_rNextColor.m_FirstColor.a, + dwTransitionTime); + } + } } -auto CSkyBox::Refresh() -> void +void CSkyBox::SetSkyColor(const TVectorGradientColor & c_rColorVector, const TVectorGradientColor & c_rNextColorVector, long lTransitionTime) { - MT_PLATFORM_STUB(); + if (c_rColorVector.empty() || c_rNextColorVector.empty()) + return; + + unsigned long ulVectorGradientColornum = 0; + unsigned long uck; + for (unsigned char ucj = 0; ucj < 4; ++ucj) + { + TSkyObjectFace & aFace = m_Faces[ucj]; + ulVectorGradientColornum = 0; + for (uck = 0; uck < aFace.m_SkyObjectQuadVector.size(); ++uck) + { + if (ulVectorGradientColornum >= c_rColorVector.size() || ulVectorGradientColornum >= c_rNextColorVector.size()) + break; + CSkyObjectQuad & aSkyObjectQuad = aFace.m_SkyObjectQuadVector[uck]; + + aSkyObjectQuad.SetSrcColor(0, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.r, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.g, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.b, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.a); + aSkyObjectQuad.SetTransition(0, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.r, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.g, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.b, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.a, + lTransitionTime); + aSkyObjectQuad.SetSrcColor(1, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.r, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.g, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.b, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.a); + aSkyObjectQuad.SetTransition(1, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.r, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.g, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.b, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.a, + lTransitionTime); + aSkyObjectQuad.SetSrcColor(2, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.r, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.g, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.b, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.a); + aSkyObjectQuad.SetTransition(2, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.r, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.g, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.b, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.a, + lTransitionTime); + aSkyObjectQuad.SetSrcColor(3, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.r, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.g, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.b, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.a); + aSkyObjectQuad.SetTransition(3, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.r, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.g, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.b, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.a, + lTransitionTime); + + ulVectorGradientColornum++; + } + } + + TSkyObjectFace & aFaceTop = m_Faces[4]; + ulVectorGradientColornum = 0; + for (uck = 0; uck < aFaceTop.m_SkyObjectQuadVector.size(); ++uck) + { + CSkyObjectQuad & aSkyObjectQuad = aFaceTop.m_SkyObjectQuadVector[uck]; + for (unsigned char v = 0; v < 4; ++v) + { + aSkyObjectQuad.SetSrcColor(v, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.r, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.g, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.b, + c_rColorVector[ulVectorGradientColornum].m_FirstColor.a); + aSkyObjectQuad.SetTransition(v, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.r, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.g, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.b, + c_rNextColorVector[ulVectorGradientColornum].m_FirstColor.a, + lTransitionTime); + } + } + + TSkyObjectFace & aFaceBottom = m_Faces[5]; + ulVectorGradientColornum = c_rColorVector.size() - 1; + for (uck = 0; uck < aFaceBottom.m_SkyObjectQuadVector.size(); ++uck) + { + CSkyObjectQuad & aSkyObjectQuad = aFaceBottom.m_SkyObjectQuadVector[uck]; + for (unsigned char v = 0; v < 4; ++v) + { + aSkyObjectQuad.SetSrcColor(v, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.r, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.g, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.b, + c_rColorVector[ulVectorGradientColornum].m_SecondColor.a); + aSkyObjectQuad.SetTransition(v, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.r, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.g, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.b, + c_rNextColorVector[ulVectorGradientColornum].m_SecondColor.a, + lTransitionTime); + } + } } -auto CSkyBox::SetSkyColor(const TVectorGradientColor &, const TVectorGradientColor &, long) -> void +void CSkyBox::StartTransition() { - MT_PLATFORM_STUB(); + m_bTransitionStarted = true; + for (unsigned char ucj = 0; ucj < 6; ++ucj) + m_Faces[ucj].StartTransition(); + m_FaceCloud.StartTransition(); } -auto CSkyBox::StartTransition() -> void +void CSkyBox::Update() { - MT_PLATFORM_STUB(); + CSkyObject::Update(); + + if (!m_bTransitionStarted) + return; + + bool bResult = false; + for (unsigned char uci = 0; uci < 6; ++uci) + bResult = m_Faces[uci].Update() || bResult; + bResult = m_FaceCloud.Update() || bResult; + + m_bTransitionStarted = bResult; } -auto CSkyBox::SetSkyObjectQuadVertical(TSkyObjectQuadVector *, const D3DXVECTOR2 *) -> void +void CSkyBox::Render() { - MT_PLATFORM_STUB(); + if (!IsNativeTerrainRenderEnabled()) + return; + + STATEMANAGER.SaveRenderState(D3DRS_ZENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_ZWRITEENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_LIGHTING, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_FOGENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHABLENDENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + + STATEMANAGER.SaveTextureStageState(0, D3DTSS_COLOROP, D3DTOP_SELECTARG2); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_COLORARG2, D3DTA_DIFFUSE); + + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + + STATEMANAGER.SetVertexShader(D3DFVF_XYZ | D3DFVF_DIFFUSE | D3DFVF_TEX1); + + STATEMANAGER.SetTransform(D3DTS_WORLD, &m_matWorld); + + if (m_ucRenderMode == CSkyObject::SKY_RENDER_MODE_TEXTURE) + { + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_SELECTARG1); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ADDRESSU, D3DTADDRESS_CLAMP); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ADDRESSV, D3DTADDRESS_CLAMP); + + for (unsigned int i = 0; i < 6; ++i) + { + CGraphicImageInstance * pFaceImageInstance = m_GraphicImageInstanceMap[m_Faces[i].m_strFaceTextureFileName]; + if (!pFaceImageInstance) + break; + + STATEMANAGER.SetTexture(0, pFaceImageInstance->GetTextureReference().GetD3DTexture()); + m_Faces[i].Render(); + } + + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ADDRESSU); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ADDRESSV); + } + else + { + STATEMANAGER.SetTexture(0, NULL); + for (unsigned int i = 0; i < 6; ++i) + { + m_Faces[i].Render(); + } + } + + STATEMANAGER.RestoreRenderState(D3DRS_CULLMODE); + STATEMANAGER.RestoreRenderState(D3DRS_LIGHTING); + STATEMANAGER.RestoreRenderState(D3DRS_ZENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ZWRITEENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_FOGENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHABLENDENABLE); + + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_COLOROP); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_COLORARG1); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_COLORARG2); } -auto CSkyBox::SetSkyObjectQuadHorizon(TSkyObjectQuadVector *, const D3DXVECTOR3 *) -> void +void CSkyBox::RenderCloud() { - MT_PLATFORM_STUB(); + if (!IsNativeTerrainRenderEnabled()) + return; + + CGraphicImageInstance * pCloudGraphicImageInstance = m_GraphicImageInstanceMap[m_FaceCloud.m_strfacename]; + if (!pCloudGraphicImageInstance || pCloudGraphicImageInstance->IsEmpty() || !pCloudGraphicImageInstance->GetTexturePointer()) + return; + + STATEMANAGER.SaveRenderState(D3DRS_ZENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_ZWRITEENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_LIGHTING, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_FOGENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHABLENDENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_SRCBLEND, D3DBLEND_ONE); + STATEMANAGER.SaveRenderState(D3DRS_DESTBLEND, D3DBLEND_INVSRCCOLOR); + STATEMANAGER.SaveRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + + STATEMANAGER.SaveTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS, D3DTTFF_COUNT2); + + m_matTextureCloud._31 = m_fCloudPositionU; + m_matTextureCloud._32 = m_fCloudPositionV; + + DWORD dwCurTime = CTimer::Instance().GetCurrentMillisecond(); + + m_fCloudPositionU += m_fCloudScrollSpeedU * (float)(dwCurTime - m_dwlastTime) * 0.001f; + if (m_fCloudPositionU >= 1.0f) + m_fCloudPositionU = 0.0f; + + m_fCloudPositionV += m_fCloudScrollSpeedV * (float)(dwCurTime - m_dwlastTime) * 0.001f; + if (m_fCloudPositionV >= 1.0f) + m_fCloudPositionV = 0.0f; + + m_dwlastTime = dwCurTime; + + STATEMANAGER.SaveTransform(D3DTS_TEXTURE0, &m_matTextureCloud); + + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_MODULATEINVALPHA_ADDCOLOR); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG2, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_SELECTARG1); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG1, D3DTA_TEXTURE); + + D3DXMATRIX matProjCloud; + D3DXMatrixPerspectiveFovRH(&matProjCloud, D3DX_PI * 0.25f, 1.33333f, 50.0f, 999999.0f); + STATEMANAGER.SetTransform(D3DTS_WORLD, &m_matWorldCloud); + STATEMANAGER.SaveTransform(D3DTS_PROJECTION, &matProjCloud); + STATEMANAGER.SetTexture(0, pCloudGraphicImageInstance->GetTexturePointer()->GetD3DTexture()); + m_FaceCloud.Render(); + STATEMANAGER.RestoreTransform(D3DTS_PROJECTION); + + STATEMANAGER.RestoreTransform(D3DTS_TEXTURE0); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS); + + STATEMANAGER.RestoreRenderState(D3DRS_CULLMODE); + STATEMANAGER.RestoreRenderState(D3DRS_LIGHTING); + STATEMANAGER.RestoreRenderState(D3DRS_ZENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ZWRITEENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_FOGENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHABLENDENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_SRCBLEND); + STATEMANAGER.RestoreRenderState(D3DRS_DESTBLEND); } diff --git a/extension/src/platform/EterLib/UIRenderCommands.cpp b/extension/src/platform/EterLib/UIRenderCommands.cpp index 8260ebc9..f1aa71db 100644 --- a/extension/src/platform/EterLib/UIRenderCommands.cpp +++ b/extension/src/platform/EterLib/UIRenderCommands.cpp @@ -1,11 +1,13 @@ #include "EterLib/StdAfx.h" #include "UIRenderCommands.h" +#include "RenderCommands3D.h" #include "EterLib/StateManager.h" #include namespace { std::vector commands; +std::uint64_t frame_id = 0; int canvas_width = 0; int canvas_height = 0; float clip_x1 = 0, clip_y1 = 0, clip_x2 = 0, clip_y2 = 0; @@ -18,14 +20,17 @@ void UIRenderGetSize(unsigned* width, unsigned* height) { if (height) *height = canvas_height; } void UIRenderBeginFrame() { + ++frame_id; commands.clear(); clip_x1 = clip_y1 = 0; clip_x2 = canvas_width; clip_y2 = canvas_height; } +std::uint64_t UIRenderFrameId() { return frame_id; } void UIRenderAdd(UIRenderCommand command) { command.clip_x1 = clip_x1; command.clip_y1 = clip_y1; command.clip_x2 = clip_x2; command.clip_y2 = clip_y2; + command.behind_3d = Render3DDraws().empty(); commands.push_back(command); } const std::vector& UIRenderCommands() { return commands; } diff --git a/extension/src/platform/EterLib/UIRenderCommands.h b/extension/src/platform/EterLib/UIRenderCommands.h index 8217eedf..7a712ee5 100644 --- a/extension/src/platform/EterLib/UIRenderCommands.h +++ b/extension/src/platform/EterLib/UIRenderCommands.h @@ -23,6 +23,7 @@ struct UIRenderCommand { // circular minimap_image_filter): the mask's texture coordinates at the quad's four corners. std::string mask; float mu[4] = {}, mv[4] = {}; + bool behind_3d = false; }; // D3DXCOLOR (0..1 floats) -> the 0xAARRGGBB the commands carry. @@ -54,6 +55,8 @@ struct IDirect3DBaseTexture8; void UIRenderReleaseMemoryTexture(IDirect3DTexture8* texture); // The name (as UIRenderTextureName) of the texture a D3D handle belongs to; "" for null or unknown. std::string UIRenderTextureNameFromHandle(const IDirect3DBaseTexture8* handle); +std::string MtCpuTextureNameFromHandle(const IDirect3DBaseTexture8* handle); +bool MtCpuMemoryTexture(const std::string& name, UIMemoryTexture* out); // A textured quad from 40250's TPDTVertex[4] (TL, TR, BL, BR) positions and texture coordinates. // PORT: D3D8 puts pixel centres on integers, which is why 40250 subtracts 0.5 from every vertex; the @@ -64,6 +67,7 @@ void UIRenderAddImage(const CGraphicTexture* texture, const float x[4], const fl void UIRenderSetSize(int width, int height); void UIRenderGetSize(unsigned* width, unsigned* height); void UIRenderBeginFrame(); +std::uint64_t UIRenderFrameId(); void UIRenderAdd(UIRenderCommand command); const std::vector& UIRenderCommands(); void UIRenderSetClip(float x, float y, float width, float height); diff --git a/extension/src/platform/EterPythonLib/PythonGraphic.cpp b/extension/src/platform/EterPythonLib/PythonGraphic.cpp index d386c441..5abf0d0d 100644 --- a/extension/src/platform/EterPythonLib/PythonGraphic.cpp +++ b/extension/src/platform/EterPythonLib/PythonGraphic.cpp @@ -78,11 +78,19 @@ LPDIRECT3D8 CPythonGraphic::GetD3D() void CPythonGraphic::SetViewport(float x, float y, float width, float height) { + if (ms_lpd3dDevice) + { + ms_lpd3dDevice->GetViewport(&m_backupViewport); + D3DVIEWPORT8 vp = { (DWORD)x, (DWORD)y, (DWORD)width, (DWORD)height, 0.0f, 1.0f }; + ms_lpd3dDevice->SetViewport(&vp); + } UIRenderSetClip(x, y, width, height); } void CPythonGraphic::RestoreViewport() { + if (ms_lpd3dDevice) + ms_lpd3dDevice->SetViewport(&m_backupViewport); UIRenderRestoreClip(); } void CPythonGraphic::SetOmniLight() diff --git a/extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp b/extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp index d86415d6..277fee42 100644 --- a/extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp +++ b/extension/src/platform/GameLib/MapOutdoorCharacterShadow.cpp @@ -1,7 +1,4 @@ -// Platform implementation of 40250 GameLib/MapOutdoorCharacterShadow.cpp. The render-target lifecycle is -// the 40250 code as is; the shadow-map pass is not run (BeginRenderCharacterShadowToTexture reports it -// cannot render) because the terrain that samples the map is drawn by Godot, which casts its own -// character shadows (PORT-PLAN §3). +// Platform implementation of 40250 GameLib/MapOutdoorCharacterShadow.cpp. #include "GameLib/StdAfx.h" #include "EterLib/StateManager.h" #include "EterLib/Camera.h" @@ -66,14 +63,63 @@ void CMapOutdoor::ReleaseCharacterShadowTexture() SAFE_RELEASE(m_lpCharacterShadowMapTexture); } -// PORT: no shadow-map pass (see the file comment); CPythonBackground::RenderCharacterShadowToTexture -// then skips the shadow instances. +static DWORD dwLightEnable = FALSE; +static bool shadowPassBegun = false; + bool CMapOutdoor::BeginRenderCharacterShadowToTexture() { - return false; + CCamera* camera = CCameraManager::Instance().GetCurrentCamera(); + if (!camera) + return false; + if (recreate) + { + CreateCharacterShadowTexture(); + recreate = false; + } + if (!m_lpCharacterShadowMapRenderTargetSurface || !m_lpCharacterShadowMapDepthSurface) + return false; + shadowPassBegun = true; + + const D3DXVECTOR3 target = camera->GetTarget(); + const D3DXVECTOR3 eye(target.x - 1.732f * 1250.0f, + target.y - 1250.0f, target.z + 2.0f * 1.732f * 1250.0f); + const D3DXVECTOR3 up(0.0f, 0.0f, 1.0f); + D3DXMATRIX lightView, lightProj; + D3DXMatrixLookAtRH(&lightView, &eye, &target, &up); + D3DXMatrixOrthoRH(&lightProj, 2550.0f, 2550.0f, 1.0f, 15000.0f); + STATEMANAGER.SaveTransform(D3DTS_VIEW, &lightView); + STATEMANAGER.SaveTransform(D3DTS_PROJECTION, &lightProj); + dwLightEnable = STATEMANAGER.GetRenderState(D3DRS_LIGHTING); + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_TEXTUREFACTOR, 0xFF808080); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_SELECTARG1); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TFACTOR); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + + bool success = SUCCEEDED(ms_lpd3dDevice->GetRenderTarget(&m_lpBackupRenderTargetSurface)); + success = SUCCEEDED(ms_lpd3dDevice->GetDepthStencilSurface(&m_lpBackupDepthSurface)) && success; + success = SUCCEEDED(ms_lpd3dDevice->SetRenderTarget(m_lpCharacterShadowMapRenderTargetSurface, + m_lpCharacterShadowMapDepthSurface)) && success; + success = SUCCEEDED(ms_lpd3dDevice->Clear(0, NULL, D3DCLEAR_TARGET | D3DCLEAR_ZBUFFER, + 0xFFFFFFFFu, 1.0f, 0)) && success; + success = SUCCEEDED(ms_lpd3dDevice->GetViewport(&m_BackupViewport)) && success; + success = SUCCEEDED(ms_lpd3dDevice->SetViewport(&m_ShadowMapViewport)) && success; + return success; } -// PORT: nothing to restore when the pass did not begin. void CMapOutdoor::EndRenderCharacterShadowToTexture() { + if (!shadowPassBegun) + return; + shadowPassBegun = false; + ms_lpd3dDevice->SetViewport(&m_BackupViewport); + if (m_lpBackupRenderTargetSurface && m_lpBackupDepthSurface) + ms_lpd3dDevice->SetRenderTarget(m_lpBackupRenderTargetSurface, m_lpBackupDepthSurface); + SAFE_RELEASE(m_lpBackupRenderTargetSurface); + SAFE_RELEASE(m_lpBackupDepthSurface); + STATEMANAGER.RestoreTransform(D3DTS_VIEW); + STATEMANAGER.RestoreTransform(D3DTS_PROJECTION); + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, dwLightEnable); + STATEMANAGER.RestoreRenderState(D3DRS_TEXTUREFACTOR); } diff --git a/extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp b/extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp index b3cc3f8d..52cf19ac 100644 --- a/extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp +++ b/extension/src/platform/GameLib/MapOutdoorRenderHTP.cpp @@ -1,10 +1,315 @@ // Platform implementation of 40250 GameLib/MapOutdoorRenderHTP.cpp (hardware-transform terrain patches). -// PORT: the Godot Metin2World adapter draws the terrain from the same .raw/.tga/tile data (PORT-PLAN §3): -// the recorder has no texture-coordinate generation or texture transforms, so the patch splat passes are -// not replayed through the D3D8 device. +// PORT: the Godot Metin2World adapter draws the terrain from the same .raw/.tga/tile data (PORT-PLAN §3). +// When IsNativeTerrainRenderEnabled() is true (MT_NATIVE_TERRAIN=1 or standalone native renderer), +// terrain patches are emitted through CStateManager with D3DTSS_TCI_CAMERASPACEPOSITION texture transforms. #include "GameLib/StdAfx.h" #include "GameLib/MapOutdoor.h" +#include "GameLib/TerrainPatch.h" +#include "GameLib/AreaTerrain.h" +#include "EterLib/StateManager.h" +#include "../EterLib/RenderCommands3D.h" +#include void CMapOutdoor::__RenderTerrain_RenderHardwareTransformPatch() { + if (!IsNativeTerrainRenderEnabled()) + return; + const DWORD fogColor = mc_pEnvironmentData ? DWORD(mc_pEnvironmentData->FogColor) : 0xffffffff; + const float fogNear = mc_pEnvironmentData ? mc_pEnvironmentData->GetFogNearDistance() : 5000.0f; + const float fogFar = mc_pEnvironmentData ? mc_pEnvironmentData->GetFogFarDistance() : 10000.0f; + + m_matWorldForCommonUse._41 = 0.0f; + m_matWorldForCommonUse._42 = 0.0f; + STATEMANAGER.SetTransform(D3DTS_WORLD, &m_matWorldForCommonUse); + + STATEMANAGER.SaveTextureStageState(0, D3DTSS_TEXCOORDINDEX, D3DTSS_TCI_CAMERASPACEPOSITION); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS, D3DTTFF_COUNT2); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_COLORARG2, D3DTA_CURRENT); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_COLOROP, D3DTOP_MODULATE); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ALPHAARG1, D3DTA_TEXTURE); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_SELECTARG1); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ADDRESSU, D3DTADDRESS_WRAP); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ADDRESSV, D3DTADDRESS_WRAP); + + STATEMANAGER.SaveTextureStageState(1, D3DTSS_TEXCOORDINDEX, D3DTSS_TCI_CAMERASPACEPOSITION); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_TEXTURETRANSFORMFLAGS, D3DTTFF_COUNT2); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_COLORARG1, D3DTA_CURRENT); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_ALPHAARG1, D3DTA_TEXTURE); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_ADDRESSU, D3DTADDRESS_CLAMP); + STATEMANAGER.SaveTextureStageState(1, D3DTSS_ADDRESSV, D3DTADDRESS_CLAMP); + STATEMANAGER.SetTexture(1, NULL); + + STATEMANAGER.SaveRenderState(D3DRS_ALPHABLENDENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHATESTENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAREF, 0); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAFUNC, D3DCMP_GREATER); + STATEMANAGER.SaveRenderState(D3DRS_TEXTUREFACTOR, fogColor); + STATEMANAGER.SaveRenderState(D3DRS_SRCBLEND, D3DBLEND_SRCALPHA); + STATEMANAGER.SaveRenderState(D3DRS_DESTBLEND, D3DBLEND_INVSRCALPHA); + STATEMANAGER.SaveRenderState(D3DRS_ZWRITEENABLE, TRUE); + + STATEMANAGER.SetVertexShader(D3DFVF_XYZ | D3DFVF_NORMAL); + + m_iRenderedSplatNumSqSum = 0; + m_iRenderedPatchNum = 0; + m_iRenderedSplatNum = 0; + m_RenderedTextureNumVector.clear(); + + WORD wPrimitiveCount = 0; + D3DPRIMITIVETYPE ePrimitiveType = D3DPT_TRIANGLESTRIP; + SelectIndexBuffer(0, &wPrimitiveCount, &ePrimitiveType); + const auto nearIt = std::upper_bound(m_PatchVector.begin(), m_PatchVector.end(), + std::pair(fogNear - 3200.0f, 0)); + const auto farIt = std::upper_bound(m_PatchVector.begin(), m_PatchVector.end(), + std::pair(fogFar + 1600.0f, 0)); + const float lod1 = __GetNoFogDistance(); + const float lod2 = __GetFogDistance(); + BYTE lod = 0; + auto updateLod = [&](float distance) { + if (lod == 0 && lod1 <= distance) { + lod = 1; + SelectIndexBuffer(1, &wPrimitiveCount, &ePrimitiveType); + } else if (lod == 1 && lod2 <= distance) { + lod = 2; + SelectIndexBuffer(2, &wPrimitiveCount, &ePrimitiveType); + } + }; + const DWORD fogEnabled = STATEMANAGER.GetRenderState(D3DRS_FOGENABLE); + STATEMANAGER.SetRenderState(D3DRS_FOGENABLE, FALSE); + for (auto it = m_PatchVector.begin(); it != nearIt; ++it) { + updateLod(it->first); + __HardwareTransformPatch_RenderPatchSplat(it->second, wPrimitiveCount, ePrimitiveType); + if (m_iRenderedSplatNum >= m_iSplatLimit) break; + if (m_bDrawWireFrame) DrawWireFrame(it->second, wPrimitiveCount, ePrimitiveType); + } + STATEMANAGER.SetRenderState(D3DRS_FOGENABLE, fogEnabled); + if (m_iRenderedSplatNum < m_iSplatLimit) { + for (auto it = nearIt; it != farIt; ++it) { + updateLod(it->first); + __HardwareTransformPatch_RenderPatchSplat(it->second, wPrimitiveCount, ePrimitiveType); + if (m_iRenderedSplatNum >= m_iSplatLimit) break; + if (m_bDrawWireFrame) DrawWireFrame(it->second, wPrimitiveCount, ePrimitiveType); + } + } + STATEMANAGER.SetRenderState(D3DRS_FOGENABLE, FALSE); + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, FALSE); + STATEMANAGER.SetTexture(0, NULL); + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS, FALSE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_TEXTURETRANSFORMFLAGS, FALSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TFACTOR); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_SELECTARG1); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + if (m_iRenderedSplatNum < m_iSplatLimit) { + for (auto it = farIt; it != m_PatchVector.end(); ++it) { + updateLod(it->first); + __HardwareTransformPatch_RenderPatchNone(it->second, wPrimitiveCount, ePrimitiveType); + if (m_iRenderedSplatNum >= m_iSplatLimit) break; + if (m_bDrawWireFrame) DrawWireFrame(it->second, wPrimitiveCount, ePrimitiveType); + } + } + STATEMANAGER.SetRenderState(D3DRS_FOGENABLE, fogEnabled); + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, TRUE); + std::sort(m_RenderedTextureNumVector.begin(), m_RenderedTextureNumVector.end()); + + m_matWorldForCommonUse._41 = 0.0f; + m_matWorldForCommonUse._42 = 0.0f; + STATEMANAGER.SetTexture(1, NULL); + + STATEMANAGER.RestoreRenderState(D3DRS_ZWRITEENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_DESTBLEND); + STATEMANAGER.RestoreRenderState(D3DRS_SRCBLEND); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHATESTENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAREF); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAFUNC); + STATEMANAGER.RestoreRenderState(D3DRS_TEXTUREFACTOR); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHABLENDENABLE); + + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_ADDRESSV); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_ADDRESSU); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_ALPHAOP); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_ALPHAARG1); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_COLOROP); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_COLORARG1); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_TEXTURETRANSFORMFLAGS); + STATEMANAGER.RestoreTextureStageState(1, D3DTSS_TEXCOORDINDEX); + + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ADDRESSV); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ADDRESSU); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ALPHAOP); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ALPHAARG1); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_COLOROP); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_COLORARG2); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_COLORARG1); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_TEXCOORDINDEX); +} + +void CMapOutdoor::__HardwareTransformPatch_RenderPatchNone(long patchnum, WORD wPrimitiveCount, D3DPRIMITIVETYPE ePrimitiveType) +{ + assert(NULL != m_pTerrainPatchProxyList); + CTerrainPatchProxy* patch = &m_pTerrainPatchProxyList[patchnum]; + if (!patch->isUsed()) return; + CGraphicVertexBuffer* vb = patch->HardwareTransformPatch_GetVertexBufferPtr(); + if (!vb) return; + STATEMANAGER.SetStreamSource(0, vb->GetD3DVertexBuffer(), m_iPatchTerrainVertexSize); + STATEMANAGER.DrawIndexedPrimitive(ePrimitiveType, 0, m_iPatchTerrainVertexCount, 0, wPrimitiveCount); +} + +void CMapOutdoor::__HardwareTransformPatch_RenderPatchSplat(long patchnum, WORD wPrimitiveCount, D3DPRIMITIVETYPE ePrimitiveType) +{ + assert(NULL != m_pTerrainPatchProxyList && "__HardwareTransformPatch_RenderPatchSplat"); + CTerrainPatchProxy * pTerrainPatchProxy = &m_pTerrainPatchProxyList[patchnum]; + if (!pTerrainPatchProxy->isUsed()) + return; + + long sPatchNum = pTerrainPatchProxy->GetPatchNum(); + if (sPatchNum < 0) + return; + + BYTE ucTerrainNum = pTerrainPatchProxy->GetTerrainNum(); + if (0xFF == ucTerrainNum) + return; + + CTerrain * pTerrain; + if (!GetTerrainPointer(ucTerrainNum, &pTerrain)) + return; + + CGraphicVertexBuffer* pkVB = pTerrainPatchProxy->HardwareTransformPatch_GetVertexBufferPtr(); + if (!pkVB) + return; + + WORD wCoordX, wCoordY; + pTerrain->GetCoordinate(&wCoordX, &wCoordY); + m_matWorldForCommonUse._41 = -(float)(wCoordX * CTerrainImpl::XSIZE * CTerrainImpl::CELLSCALE); + m_matWorldForCommonUse._42 = (float)(wCoordY * CTerrainImpl::YSIZE * CTerrainImpl::CELLSCALE); + + D3DXMATRIX matTerrainTexTransform, matSplatAlphaTexTransform; + D3DXMatrixMultiply(&matTerrainTexTransform, &m_matViewInverse, &m_matWorldForCommonUse); + D3DXMatrixMultiply(&matSplatAlphaTexTransform, &matTerrainTexTransform, &m_matSplatAlpha); + STATEMANAGER.SetTransform(D3DTS_TEXTURE1, &matSplatAlphaTexTransform); + + STATEMANAGER.SetStreamSource(0, pkVB->GetD3DVertexBuffer(), m_iPatchTerrainVertexSize); + // 40250 renders the terrain splats unlit; only its shadow pass enables lighting. + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, FALSE); + + TTerrainSplatPatch & rTerrainSplatPatch = pTerrain->GetTerrainSplatPatch(); + const DWORD texCount = m_TextureSet.GetTextureCount(); + bool isFirst = true; + int renderedSplatCount = 0; + + for (DWORD j = 1; j < texCount; ++j) + { + TTerainSplat & rSplat = rTerrainSplatPatch.Splats[j]; + if (!rSplat.Active || rTerrainSplatPatch.PatchTileCount[sPatchNum][j] == 0) + continue; + + const TTerrainTexture & rTexture = m_TextureSet.GetTexture(j); + if (!rTexture.pd3dTexture) + continue; + + D3DXMATRIX matTexTransform; + D3DXMatrixMultiply(&matTexTransform, &m_matViewInverse, &rTexture.m_matTransform); + STATEMANAGER.SetTransform(D3DTS_TEXTURE0, &matTexTransform); + STATEMANAGER.SetTexture(0, rTexture.pd3dTexture); + + if (isFirst || !rSplat.pd3dTexture) + { + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + STATEMANAGER.SetRenderState(D3DRS_ALPHABLENDENABLE, FALSE); + STATEMANAGER.SetRenderState(D3DRS_ZWRITEENABLE, TRUE); + isFirst = false; + } + else + { + STATEMANAGER.SetTexture(1, rSplat.pd3dTexture); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_SELECTARG1); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_SELECTARG1); + STATEMANAGER.SetRenderState(D3DRS_ALPHABLENDENABLE, TRUE); + STATEMANAGER.SetRenderState(D3DRS_ZWRITEENABLE, FALSE); + } + + STATEMANAGER.DrawIndexedPrimitive(ePrimitiveType, 0, m_iPatchTerrainVertexCount, 0, wPrimitiveCount); + ++renderedSplatCount; + ++m_iRenderedSplatNum; + if (std::find(m_RenderedTextureNumVector.begin(), m_RenderedTextureNumVector.end(), int(j)) == m_RenderedTextureNumVector.end()) + m_RenderedTextureNumVector.push_back(int(j)); + if (m_iRenderedSplatNum >= m_iSplatLimit) + break; + } + + if (renderedSplatCount == 0 && texCount > 1) + { + const TTerrainTexture & rTexture = m_TextureSet.GetTexture(1); + if (rTexture.pd3dTexture) + { + D3DXMATRIX matTexTransform; + D3DXMatrixMultiply(&matTexTransform, &m_matViewInverse, &rTexture.m_matTransform); + STATEMANAGER.SetTransform(D3DTS_TEXTURE0, &matTexTransform); + STATEMANAGER.SetTexture(0, rTexture.pd3dTexture); + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + STATEMANAGER.SetRenderState(D3DRS_ALPHABLENDENABLE, FALSE); + STATEMANAGER.SetRenderState(D3DRS_ZWRITEENABLE, TRUE); + STATEMANAGER.DrawIndexedPrimitive(ePrimitiveType, 0, m_iPatchTerrainVertexCount, 0, wPrimitiveCount); + renderedSplatCount = 1; + ++m_iRenderedSplatNum; + } + } + + if (m_bDrawShadow && pTerrain->GetShadowTexture()) + { + const DWORD previousFogColor = STATEMANAGER.GetRenderState(D3DRS_FOGCOLOR); + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, TRUE); + STATEMANAGER.SetRenderState(D3DRS_FOGCOLOR, 0xFFFFFFFF); + D3DXMATRIX matShadowTexTransform; + D3DXMatrixMultiply(&matShadowTexTransform, &matTerrainTexTransform, &m_matStaticShadow); + STATEMANAGER.SetTransform(D3DTS_TEXTURE0, &matShadowTexTransform); + STATEMANAGER.SetTexture(0, pTerrain->GetShadowTexture()); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ADDRESSU, D3DTADDRESS_CLAMP); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ADDRESSV, D3DTADDRESS_CLAMP); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + if (m_bDrawChrShadow && m_lpCharacterShadowMapTexture) + { + STATEMANAGER.SetTransform(D3DTS_TEXTURE1, &m_matDynamicShadow); + STATEMANAGER.SetTexture(1, m_lpCharacterShadowMapTexture); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLORARG2, D3DTA_CURRENT); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ADDRESSU, D3DTADDRESS_CLAMP); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ADDRESSV, D3DTADDRESS_CLAMP); + } + STATEMANAGER.SetRenderState(D3DRS_ALPHABLENDENABLE, TRUE); + STATEMANAGER.SetRenderState(D3DRS_SRCBLEND, D3DBLEND_ZERO); + STATEMANAGER.SetRenderState(D3DRS_DESTBLEND, D3DBLEND_SRCCOLOR); + STATEMANAGER.SetRenderState(D3DRS_ZWRITEENABLE, FALSE); + STATEMANAGER.DrawIndexedPrimitive(ePrimitiveType, 0, m_iPatchTerrainVertexCount, 0, wPrimitiveCount); + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ADDRESSU, D3DTADDRESS_CLAMP); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ADDRESSV, D3DTADDRESS_CLAMP); + STATEMANAGER.SetRenderState(D3DRS_SRCBLEND, D3DBLEND_SRCALPHA); + STATEMANAGER.SetRenderState(D3DRS_DESTBLEND, D3DBLEND_INVSRCALPHA); + STATEMANAGER.SetRenderState(D3DRS_FOGCOLOR, previousFogColor); + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, FALSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_SELECTARG1); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ADDRESSU, D3DTADDRESS_WRAP); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ADDRESSV, D3DTADDRESS_WRAP); + ++renderedSplatCount; + ++m_iRenderedSplatNum; + } + + ++m_iRenderedPatchNum; + m_iRenderedSplatNumSqSum += renderedSplatCount * renderedSplatCount; } diff --git a/extension/src/platform/GameLib/MapOutdoorWater.cpp b/extension/src/platform/GameLib/MapOutdoorWater.cpp index 016ff8e5..41f6f71b 100644 --- a/extension/src/platform/GameLib/MapOutdoorWater.cpp +++ b/extension/src/platform/GameLib/MapOutdoorWater.cpp @@ -3,6 +3,8 @@ #include "GameLib/StdAfx.h" #include "EterLib/StateManager.h" #include "EterLib/ResourceManager.h" +#include "EterBase/Timer.h" +#include "../EterLib/RenderCommands3D.h" #include "GameLib/MapOutdoor.h" #include "GameLib/TerrainPatch.h" @@ -24,7 +26,133 @@ void CMapOutdoor::UnloadWaterTexture() m_WaterInstances[i].Destroy(); } -// PORT: the water patches are drawn by the Godot side. void CMapOutdoor::RenderWater() { + if (!IsNativeTerrainRenderEnabled()) + return; + + if (m_PatchVector.empty()) + return; + + if (!IsVisiblePart(PART_WATER)) + return; + + D3DXMATRIX matTexTransformWater; + + STATEMANAGER.SaveRenderState(D3DRS_ZWRITEENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHABLENDENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + STATEMANAGER.SaveRenderState(D3DRS_DIFFUSEMATERIALSOURCE, D3DMCS_COLOR1); + STATEMANAGER.SaveRenderState(D3DRS_COLORVERTEX, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_LIGHTING, FALSE); + + CGraphicImageInstance& rkWaterInst = m_WaterInstances[((ELTimer_GetMSec() / 70) % 30)]; + if (!rkWaterInst.IsEmpty() && rkWaterInst.GetTexturePointer()) + STATEMANAGER.SetTexture(0, rkWaterInst.GetTexturePointer()->GetD3DTexture()); + else + STATEMANAGER.SetTexture(0, NULL); + + D3DXMatrixScaling(&matTexTransformWater, m_fWaterTexCoordBase, -m_fWaterTexCoordBase, 0.0f); + D3DXMatrixMultiply(&matTexTransformWater, &m_matViewInverse, &matTexTransformWater); + + STATEMANAGER.SaveTransform(D3DTS_TEXTURE0, &matTexTransformWater); + STATEMANAGER.SaveVertexShader(D3DFVF_XYZ | D3DFVF_DIFFUSE); + + STATEMANAGER.SaveTextureStageState(0, D3DTSS_TEXCOORDINDEX, D3DTSS_TCI_CAMERASPACEPOSITION); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS, D3DTTFF_COUNT2); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_MINFILTER, D3DTEXF_LINEAR); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_MAGFILTER, D3DTEXF_LINEAR); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_MIPFILTER, D3DTEXF_LINEAR); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ADDRESSU, D3DTADDRESS_WRAP); + STATEMANAGER.SaveTextureStageState(0, D3DTSS_ADDRESSV, D3DTADDRESS_WRAP); + + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_SELECTARG1); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG1, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_SELECTARG1); + + STATEMANAGER.SetTexture(1, NULL); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + + static float s_fWaterHeightCurrent = 0; + static float s_fWaterHeightBegin = 0; + static float s_fWaterHeightEnd = 0; + static DWORD s_dwLastHeightChangeTime = CTimer::Instance().GetCurrentMillisecond(); + static DWORD s_dwBlendtime = 300; + + if ((CTimer::Instance().GetCurrentMillisecond() - s_dwLastHeightChangeTime) > s_dwBlendtime) + { + s_dwBlendtime = random_range(1000, 3000); + + if (s_fWaterHeightEnd == 0) + s_fWaterHeightEnd = -static_cast(random_range(0, 15)); + else + s_fWaterHeightEnd = 0; + + s_fWaterHeightBegin = s_fWaterHeightCurrent; + s_dwLastHeightChangeTime = CTimer::Instance().GetCurrentMillisecond(); + } + + s_fWaterHeightCurrent = s_fWaterHeightBegin + (s_fWaterHeightEnd - s_fWaterHeightBegin) * (float)((CTimer::Instance().GetCurrentMillisecond() - s_dwLastHeightChangeTime) / (float)s_dwBlendtime); + m_matWorldForCommonUse._43 = s_fWaterHeightCurrent; + + m_matWorldForCommonUse._41 = 0.0f; + m_matWorldForCommonUse._42 = 0.0f; + STATEMANAGER.SetTransform(D3DTS_WORLD, &m_matWorldForCommonUse); + + for (auto i = m_PatchVector.begin(); i != m_PatchVector.end(); ++i) + { + DrawWater(i->second); + } + + m_matWorldForCommonUse._43 = 0.0f; + + STATEMANAGER.RestoreVertexShader(); + STATEMANAGER.RestoreTransform(D3DTS_TEXTURE0); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_MINFILTER); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_MAGFILTER); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_MIPFILTER); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ADDRESSU); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_ADDRESSV); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_TEXCOORDINDEX); + STATEMANAGER.RestoreTextureStageState(0, D3DTSS_TEXTURETRANSFORMFLAGS); + + STATEMANAGER.RestoreRenderState(D3DRS_LIGHTING); + STATEMANAGER.RestoreRenderState(D3DRS_DIFFUSEMATERIALSOURCE); + STATEMANAGER.RestoreRenderState(D3DRS_COLORVERTEX); + STATEMANAGER.RestoreRenderState(D3DRS_ZWRITEENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHABLENDENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_CULLMODE); +} + +void CMapOutdoor::DrawWater(long patchnum) +{ + assert(NULL != m_pTerrainPatchProxyList); + if (!m_pTerrainPatchProxyList) + return; + + CTerrainPatchProxy& rkTerrainPatchProxy = m_pTerrainPatchProxyList[patchnum]; + + if (!rkTerrainPatchProxy.isUsed()) + return; + + if (!rkTerrainPatchProxy.isWaterExists()) + return; + + CGraphicVertexBuffer* pkVB = rkTerrainPatchProxy.GetWaterVertexBufferPointer(); + if (!pkVB) + return; + + if (!pkVB->GetD3DVertexBuffer()) + return; + + UINT uPriCount = rkTerrainPatchProxy.GetWaterFaceCount(); + if (!uPriCount) + return; + + STATEMANAGER.SetStreamSource(0, pkVB->GetD3DVertexBuffer(), sizeof(SWaterVertex)); + STATEMANAGER.DrawPrimitive(D3DPT_TRIANGLELIST, 0, uPriCount); + + ms_faceCount += uPriCount; } diff --git a/extension/src/platform/ScriptLib/PythonBoot.cpp b/extension/src/platform/ScriptLib/PythonBoot.cpp index 41f8cdcb..0b80e3ed 100644 --- a/extension/src/platform/ScriptLib/PythonBoot.cpp +++ b/extension/src/platform/ScriptLib/PythonBoot.cpp @@ -8,6 +8,7 @@ #include "EterBase/Timer.h" #include "EterLib/Util.h" #include "EterLib/GrpBase.h" +#include "EterLib/Camera.h" #include "../EterLib/HostCursor.h" #include "UserInterface/PythonNetworkStream.h" #include "UserInterface/AccountConnector.h" @@ -86,7 +87,11 @@ std::unique_ptr g_effect_manager; class GameApplicationAdapter final : public IAbstractApplication { public: - void GetMousePosition(POINT* point) override { if (point) *point = {}; } + void GetMousePosition(POINT* point) override + { + if (point) + CPythonApplication::Instance().GetMousePosition(point); + } float GetGlobalTime() override { return CTimer::Instance().GetCurrentSecond(); } float GetGlobalElapsedTime() override { return CTimer::Instance().GetElapsedSecond(); } void SkipRenderBuffering(DWORD) override {} @@ -144,6 +149,7 @@ struct GameSingletons CPythonSystem pySystem; // last CPythonApplication member (m_pySystem) }; std::unique_ptr g_game_singletons; +bool g_host_hardware_cursor_enabled = false; bool fail(std::string* error, const std::string& text) { @@ -229,6 +235,26 @@ void init_2v0_gameplay_stubs(); namespace PythonBoot { +void SetHostHardwareCursorEnabled(bool enabled) +{ + g_host_hardware_cursor_enabled = enabled; +} + +bool HostHardwareCursorEnabled() +{ + return g_host_hardware_cursor_enabled; +} + +int CursorShape() +{ + return g_game_singletons ? CPythonApplication::Instance().GetCursorNum() : 0; +} + +bool CursorVisible() +{ + return !g_game_singletons || CPythonApplication::Instance().GetCursorVisible(); +} + std::string CurrentMapName() { if (!g_game_singletons) @@ -255,9 +281,43 @@ void SetUISize(int width, int height) } } +void update_3d_pick_ray() +{ + if (!g_game_singletons || !g_window_manager) + return; + if (!CPythonBackground::Instance().IsMapReady()) + return; + long lx = 0, ly = 0; + g_window_manager->GetMousePosition(lx, ly); + const long sw = g_window_manager->GetScreenWidth(); + const long sh = g_window_manager->GetScreenHeight(); + if (sw <= 0 || sh <= 0) + return; + CScreen s; + s.SetPerspective(30.0f, g_window_manager->GetAspect(), 100.0f, CPythonBackground::Instance().GetFarClip()); + s.UpdateViewMatrix(); + s.BuildViewFrustum(); + s.SetCursorPosition(lx, ly, sw, sh); + POINT ptMouse{static_cast(lx), static_cast(ly)}; + CPythonItem::Instance().Update(ptMouse); + CPythonCharacterManager::Instance().Pick(); +} + void UIMouseMove(int x, int y) { MtHostSetCursor(x, y); + if (g_game_singletons) + { + CPythonApplication::Instance().OnMouseMove(x, y); + if (g_window_manager) + { + long lx = x, ly = y; + g_window_manager->GetMousePosition(lx, ly); + MtHostSetCursor(static_cast(lx), static_cast(ly)); + } + update_3d_pick_ray(); + return; + } if (g_window_manager) g_window_manager->RunMouseMove(x, y); } @@ -265,16 +325,55 @@ void UIMouseButton(int button, bool pressed, int x, int y) { MtHostSetCursor(x, y); if (!g_window_manager) return; - g_window_manager->RunMouseMove(x, y); + if (g_game_singletons) + CPythonApplication::Instance().OnMouseMove(x, y); + else + g_window_manager->RunMouseMove(x, y); + update_3d_pick_ray(); switch (button) { - case 1: if (pressed) g_window_manager->RunMouseLeftButtonDown(x, y); else g_window_manager->RunMouseLeftButtonUp(x, y); break; - case 2: if (pressed) g_window_manager->RunMouseRightButtonDown(x, y); else g_window_manager->RunMouseRightButtonUp(x, y); break; - case 3: if (pressed) g_window_manager->RunMouseMiddleButtonDown(x, y); else g_window_manager->RunMouseMiddleButtonUp(x, y); break; + case 1: + if (pressed) + g_window_manager->RunMouseLeftButtonDown(x, y); + else + g_window_manager->RunMouseLeftButtonUp(x, y); + break; + case 2: + if (pressed) + { + g_window_manager->RunMouseRightButtonDown(x, y); + } + else + { + g_window_manager->RunMouseRightButtonUp(x, y); + if (g_game_singletons) + { + CCamera* pkCmrCur = CCameraManager::Instance().GetCurrentCamera(); + if (pkCmrCur && pkCmrCur->IsDraging()) + CPythonApplication::Instance().OnMouseMiddleButtonUp(x, y); + } + } + break; + case 3: + if (g_game_singletons) + { + if (pressed) CPythonApplication::Instance().OnMouseMiddleButtonDown(x, y); + else CPythonApplication::Instance().OnMouseMiddleButtonUp(x, y); + } + if (pressed) g_window_manager->RunMouseMiddleButtonDown(x, y); else g_window_manager->RunMouseMiddleButtonUp(x, y); + break; default: break; } } +void UIMouseWheel(int nLen) +{ + CCameraManager& rkCmrMgr = CCameraManager::Instance(); + CCamera* pkCmrCur = rkCmrMgr.GetCurrentCamera(); + if (pkCmrCur) + pkCmrCur->Wheel(nLen); +} + // 40250 CPythonApplication::OnKeyDown/OnKeyUp (PythonApplicationEvent.cpp:130-146): ESC first goes // to RunPressEscapeKey (the OnPressEscapeKey chain every dialog closes on), then to RunKeyDown. void UIKey(int key, bool pressed) @@ -282,12 +381,80 @@ void UIKey(int key, bool pressed) if (!g_window_manager) return; if (pressed) { + CPythonApplication::Instance().KeyDown(key); if (DIK_ESCAPE == key) g_window_manager->RunPressEscapeKey(); g_window_manager->RunKeyDown(key); } else + { + CPythonApplication::Instance().KeyUp(key); g_window_manager->RunKeyUp(key); + } +} + +void SetMoveDirection(float angleDeg, bool moving) +{ + if (!g_game_singletons) return; + if (moving) + CPythonPlayer::Instance().NEW_MoveToDirection(angleDeg); + else + CPythonPlayer::Instance().NEW_Stop(); +} + +void SetAttackKey(bool pressed) +{ + if (!g_game_singletons) return; + CPythonPlayer::Instance().SetAttackKeyState(pressed); +} + +void CameraBeginDrag(int x, int y) +{ + CCameraManager& rkCmrMgr = CCameraManager::Instance(); + CCamera* pkCmrCur = rkCmrMgr.GetCurrentCamera(); + if (pkCmrCur) + pkCmrCur->BeginDrag(x, y); +} + +void CameraDrag(int x, int y) +{ + CCameraManager& rkCmrMgr = CCameraManager::Instance(); + CCamera* pkCmrCur = rkCmrMgr.GetCurrentCamera(); + if (pkCmrCur) + { + POINT pt; + pkCmrCur->Drag(x, y, &pt); + } +} + +void CameraEndDrag() +{ + CCameraManager& rkCmrMgr = CCameraManager::Instance(); + CCamera* pkCmrCur = rkCmrMgr.GetCurrentCamera(); + if (pkCmrCur) + pkCmrCur->EndDrag(); +} + +bool IsPointInsideActiveUI(int x, int y) +{ + if (!g_window_manager) return false; + return g_window_manager->IsPointInsideActiveUI(x, y); +} + +PlayerStatusInfo GetPlayerStatusInfo() +{ + PlayerStatusInfo info{}; + if (!g_game_singletons) return info; + info.level = CPythonPlayer::Instance().GetStatus(POINT_LEVEL); + info.hp = CPythonPlayer::Instance().GetStatus(POINT_HP); + info.max_hp = CPythonPlayer::Instance().GetStatus(POINT_MAX_HP); + info.sp = CPythonPlayer::Instance().GetStatus(POINT_SP); + info.max_sp = CPythonPlayer::Instance().GetStatus(POINT_MAX_SP); + info.exp = CPythonPlayer::Instance().GetStatus(POINT_EXP); + info.max_exp = CPythonPlayer::Instance().GetStatus(POINT_NEXT_EXP); + const char* szName = CPythonPlayer::Instance().GetName(); + if (szName) info.name = szName; + return info; } // 40250 CPythonApplication::WindowProcedure (PythonApplicationProcedure.cpp:111-117): WM_CHAR goes to @@ -346,6 +513,14 @@ void UIRender() if (g_window_manager) g_window_manager->Render(); } +bool IsSoftwareCursorVisible() +{ + if (!g_game_singletons) + return false; + return CPythonApplication::Instance().GetCursorMode() == CPythonApplication::CURSOR_MODE_SOFTWARE && + CPythonApplication::Instance().GetCursorVisible(); +} + bool Start(const char* stdlib_path, std::string* error) { if (g_launcher) @@ -647,6 +822,16 @@ void Stop() g_fiber.finished = false; // 40250 Main(): pyLauncher.Clear() runs before app->Destroy()/delete app, so the window manager // (a CPythonApplication member) is still alive while Py_Finalize runs the windows' __del__. + if (Py_IsInitialized()) + { + PyGILState_STATE gil = PyGILState_Ensure(); + PyRun_SimpleString( + "import sys\n" + "_mm = sys.modules.get('mouseModule')\n" + "if _mm and hasattr(_mm, 'mouseController'):\n" + " getattr(_mm.mouseController, 'cursorDict', {}).clear()\n"); + PyGILState_Release(gil); + } g_launcher->Clear(); g_game_singletons.reset(); g_game_application.reset(); diff --git a/extension/src/platform/ScriptLib/PythonBoot.h b/extension/src/platform/ScriptLib/PythonBoot.h index 77503b13..54e30996 100644 --- a/extension/src/platform/ScriptLib/PythonBoot.h +++ b/extension/src/platform/ScriptLib/PythonBoot.h @@ -24,6 +24,11 @@ namespace PythonBoot // before the launcher exists. bool Start(const char* stdlib_path, std::string* error); bool IsRunning(); +// The SDL native client supplies an OS cursor; the Godot host keeps its software cursor. +void SetHostHardwareCursorEnabled(bool enabled); +bool HostHardwareCursorEnabled(); +int CursorShape(); +bool CursorVisible(); // 40250: RunMainScript(CPythonLauncher&, const char*) — the module initializers, __DEBUG__, // __COMMAND_LINE__ and RunFile("system.py"). 2V0-e registers temporary observable gameplay modules; @@ -56,13 +61,34 @@ bool Evaluate(const char* expression, std::string* result, std::string* error); void SetUISize(int width, int height); void UIMouseMove(int x, int y); void UIMouseButton(int button, bool pressed, int x, int y); +void UIMouseWheel(int nLen); void UIKey(int key, bool pressed); +// Mobile touch & joystick controls +void SetMoveDirection(float angleDeg, bool moving); +void SetAttackKey(bool pressed); +void CameraBeginDrag(int x, int y); +void CameraDrag(int x, int y); +void CameraEndDrag(); +bool IsPointInsideActiveUI(int x, int y); + +struct PlayerStatusInfo { + int level = 1; + int hp = 0; + int max_hp = 0; + int sp = 0; + int max_sp = 0; + int exp = 0; + int max_exp = 0; + std::string name; +}; +PlayerStatusInfo GetPlayerStatusInfo(); // WM_CHAR (a Unicode code point) and WM_KEYDOWN (a Win32 VK code) for CPythonIME and OnIMEKeyDown. void UIChar(unsigned codepoint); void UIIMEKeyDown(int vkey); // While IsAppLooping(), UIUpdate is AppFrame() and UIRender keeps the commands that Process() drew. void UIUpdate(); void UIRender(); +bool IsSoftwareCursorVisible(); // 40250 Main() calls Clear() and lets the launcher leave scope. void Stop(); diff --git a/extension/src/platform/SpeedTreeLib/SpeedTreeForest.cpp b/extension/src/platform/SpeedTreeLib/SpeedTreeForest.cpp index 89221b41..1f71ff9f 100644 --- a/extension/src/platform/SpeedTreeLib/SpeedTreeForest.cpp +++ b/extension/src/platform/SpeedTreeLib/SpeedTreeForest.cpp @@ -1,84 +1,241 @@ -// Platform skeleton for SpeedTreeLib/SpeedTreeForest.h (40250 SpeedTreeLib/SpeedTreeForest.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. #include "SpeedTreeLib/StdAfx.h" #include "SpeedTreeLib/SpeedTreeForest.h" -#include "../PlatformStub.h" +#include "EterPack/EterPackManager.h" +#include "GameLib/Property.h" +#include "GameLib/PropertyManager.h" +#include "../EterLib/RenderCommands3D.h" + +#include +#include +#include +#include +#include CSpeedTreeForest::CSpeedTreeForest() + : m_fWindStrength(0.2f) + , m_fAccumTime(0.0f) { - MT_PLATFORM_STUB(); + std::memset(m_afLighting, 0, sizeof(m_afLighting)); + std::memset(m_afFog, 0, sizeof(m_afFog)); + m_afForestExtents[0] = m_afForestExtents[1] = m_afForestExtents[2] = FLT_MAX; + m_afForestExtents[3] = m_afForestExtents[4] = m_afForestExtents[5] = -FLT_MAX; } CSpeedTreeForest::~CSpeedTreeForest() { - MT_PLATFORM_STUB(); } auto CSpeedTreeForest::ClearMainTree() -> void { - MT_PLATFORM_STUB(); + Clear(); } -auto CSpeedTreeForest::GetMainTree(DWORD, CSpeedTreeWrapper **, const char *) -> BOOL +auto CSpeedTreeForest::GetMainTree(DWORD dwCRC, CSpeedTreeWrapper ** ppMainTree, const char * c_pszFileName) -> BOOL { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + if (!ppMainTree) + return FALSE; + *ppMainTree = NULL; + + if (!IsNativeTerrainRenderEnabled()) + return FALSE; + + TTreeMap::iterator itor = m_pMainTreeMap.find(dwCRC); + CSpeedTreeWrapper * pTree = NULL; + + if (itor != m_pMainTreeMap.end()) + { + pTree = itor->second; + } + else + { + if (!c_pszFileName || !c_pszFileName[0]) + return FALSE; + + CMappedFile file; + LPCVOID c_pvData = NULL; + if (!CEterPackManager::Instance().Get(file, c_pszFileName, &c_pvData)) + return FALSE; + + float fSize = 1000.0f; + float fVariance = 0.0f; + CProperty * pProperty = NULL; + if (CPropertyManager::InstancePtr() && CPropertyManager::Instance().Get(dwCRC, &pProperty) && pProperty) + { + const char * c_pszTreeSize = NULL; + const char * c_pszTreeVariance = NULL; + if (pProperty->GetString("TreeSize", &c_pszTreeSize) && c_pszTreeSize) + fSize = static_cast( std::atof(c_pszTreeSize)); + if (pProperty->GetString("TreeVariance", &c_pszTreeVariance) && c_pszTreeVariance) + fVariance = static_cast(std::atof(c_pszTreeVariance)); + } + + pTree = new CSpeedTreeWrapper; + if (!pTree->LoadTree(c_pszFileName, static_cast(c_pvData), file.Size(), 1, fSize, fVariance)) + { + delete pTree; + return FALSE; + } + + m_pMainTreeMap.insert(TTreeMap::value_type(dwCRC, pTree)); + file.Destroy(); + } + + *ppMainTree = pTree; + return TRUE; } -auto CSpeedTreeForest::GetMainTree(DWORD) -> CSpeedTreeWrapper * +auto CSpeedTreeForest::GetMainTree(DWORD dwCRC) -> CSpeedTreeWrapper * { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + TTreeMap::iterator itor = m_pMainTreeMap.find(dwCRC); + if (itor == m_pMainTreeMap.end()) + return NULL; + return itor->second; } -auto CSpeedTreeForest::DeleteMainTree(DWORD) -> void +auto CSpeedTreeForest::DeleteMainTree(DWORD dwCRC) -> void { - MT_PLATFORM_STUB(); + TTreeMap::iterator itor = m_pMainTreeMap.find(dwCRC); + if (itor == m_pMainTreeMap.end()) + return; + + CSpeedTreeWrapper * pMainTree = itor->second; + UINT uiCount = 0; + CSpeedTreeWrapper ** ppInstances = pMainTree->GetInstances(uiCount); + for (UINT i = 0; i < uiCount; ++i) + delete ppInstances[i]; + + delete pMainTree; + m_pMainTreeMap.erase(itor); } -auto CSpeedTreeForest::CreateInstance(float, float, float, DWORD, const char *) -> CSpeedTreeWrapper * +auto CSpeedTreeForest::CreateInstance(float x, float y, float z, DWORD dwTreeCRC, const char * c_pszTreeName) -> CSpeedTreeWrapper * { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + if (!IsNativeTerrainRenderEnabled()) + return NULL; + + CSpeedTreeWrapper * pMainTree = NULL; + if (!GetMainTree(dwTreeCRC, &pMainTree, c_pszTreeName) || !pMainTree) + return NULL; + + CSpeedTreeWrapper * pTreeInst = pMainTree->MakeInstance(); + if (!pTreeInst) + return NULL; + + pTreeInst->SetPosition(x, y, z); + pTreeInst->RegisterBoundingSphere(); + AdjustExtents(x, y, z); + return pTreeInst; } -auto CSpeedTreeForest::DeleteInstance(CSpeedTreeWrapper *) -> void +auto CSpeedTreeForest::DeleteInstance(CSpeedTreeWrapper * pInstance) -> void { - MT_PLATFORM_STUB(); + if (!pInstance) + return; + + CSpeedTreeWrapper * pParentTree = pInstance->InstanceOf(); + if (!pParentTree) + return; + + pParentTree->DeleteInstance(pInstance); } -auto CSpeedTreeForest::UpdateSystem(float) -> void +auto CSpeedTreeForest::UpdateSystem(float fCurrentTime) -> void { - MT_PLATFORM_STUB(); + static float fLastTime = fCurrentTime; + float fElapsedTime = fCurrentTime - fLastTime; + fLastTime = fCurrentTime; + if (fElapsedTime > 0.0f && fElapsedTime < 10.0f) + m_fAccumTime += fElapsedTime; + SetupWindMatrices(m_fAccumTime); } auto CSpeedTreeForest::Clear() -> void { - MT_PLATFORM_STUB(); + TTreeMap::iterator itor = m_pMainTreeMap.begin(); + UINT uiCount = 0; + + while (itor != m_pMainTreeMap.end()) + { + CSpeedTreeWrapper * pMainTree = (itor++)->second; + if (!pMainTree) + continue; + CSpeedTreeWrapper ** ppInstances = pMainTree->GetInstances(uiCount); + for (UINT i = 0; i < uiCount; ++i) + delete ppInstances[i]; + delete pMainTree; + } + + m_pMainTreeMap.clear(); } -auto CSpeedTreeForest::SetLight(const float *, const float *, const float *) -> void +auto CSpeedTreeForest::SetLight(const float * afDirection, const float * afAmbient, const float * afDiffuse) -> void { - MT_PLATFORM_STUB(); + if (!afDirection || !afAmbient || !afDiffuse) + return; + + m_afLighting[0] = afDirection[0]; + m_afLighting[1] = afDirection[1]; + m_afLighting[2] = afDirection[2]; + m_afLighting[3] = 1.0f; + + m_afLighting[4] = afAmbient[0]; + m_afLighting[5] = afAmbient[1]; + m_afLighting[6] = afAmbient[2]; + m_afLighting[7] = afAmbient[3]; + + m_afLighting[8] = afDiffuse[0]; + m_afLighting[9] = afDiffuse[1]; + m_afLighting[10] = afDiffuse[2]; + m_afLighting[11] = afDiffuse[3]; } -auto CSpeedTreeForest::SetFog(float, float) -> void +auto CSpeedTreeForest::SetFog(float fFogNear, float fFogFar) -> void { - MT_PLATFORM_STUB(); + const float denom = (fFogFar - fFogNear); + const float c_fFogLinearScale = (std::fabs(denom) > 1e-4f) ? (1.0f / denom) : 0.0f; + + m_afFog[0] = fFogNear; + m_afFog[1] = fFogFar; + m_afFog[2] = c_fFogLinearScale; + m_afFog[3] = 0.0f; } -auto CSpeedTreeForest::SetWindStrength(float) -> void +auto CSpeedTreeForest::SetWindStrength(float fStrength) -> void { - MT_PLATFORM_STUB(); + if (m_fWindStrength == fStrength) + return; + + m_fWindStrength = fStrength; + + TTreeMap::iterator itor = m_pMainTreeMap.begin(); + UINT uiCount = 0; + + while (itor != m_pMainTreeMap.end()) + { + CSpeedTreeWrapper * pMainTree = (itor++)->second; + if (!pMainTree) + continue; + CSpeedTreeWrapper ** ppInstances = pMainTree->GetInstances(uiCount); + for (UINT i = 0; i < uiCount; ++i) + { + if (ppInstances[i] && ppInstances[i]->GetSpeedTree()) + ppInstances[i]->GetSpeedTree()->SetWindStrength(m_fWindStrength); + } + } } -auto CSpeedTreeForest::SetupWindMatrices(float) -> void +auto CSpeedTreeForest::SetupWindMatrices(float /*fTimeInSecs*/) -> void { - MT_PLATFORM_STUB(); } -auto CSpeedTreeForest::AdjustExtents(float, float, float) -> void +auto CSpeedTreeForest::AdjustExtents(float x, float y, float z) -> void { - MT_PLATFORM_STUB(); + m_afForestExtents[0] = std::min(m_afForestExtents[0], x); + m_afForestExtents[1] = std::min(m_afForestExtents[1], y); + m_afForestExtents[2] = std::min(m_afForestExtents[2], z); + + m_afForestExtents[3] = std::max(m_afForestExtents[3], x); + m_afForestExtents[4] = std::max(m_afForestExtents[4], y); + m_afForestExtents[5] = std::max(m_afForestExtents[5], z); } diff --git a/extension/src/platform/SpeedTreeLib/SpeedTreeForestDirectX8.cpp b/extension/src/platform/SpeedTreeLib/SpeedTreeForestDirectX8.cpp index 5bf07b86..f657e8fe 100644 --- a/extension/src/platform/SpeedTreeLib/SpeedTreeForestDirectX8.cpp +++ b/extension/src/platform/SpeedTreeLib/SpeedTreeForestDirectX8.cpp @@ -1,43 +1,163 @@ -// Platform skeleton for SpeedTreeLib/SpeedTreeForestDirectX8.h (40250 SpeedTreeLib/SpeedTreeForestDirectX8.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. #include "SpeedTreeLib/StdAfx.h" #include "SpeedTreeLib/SpeedTreeForestDirectX8.h" -#include "../PlatformStub.h" +#include "EterBase/Timer.h" +#include "EterLib/Camera.h" +#include "EterLib/StateManager.h" +#include "../EterLib/RenderCommands3D.h" CSpeedTreeForestDirectX8::CSpeedTreeForestDirectX8() + : m_pDx(NULL) + , m_dwBranchVertexShader(D3DFVF_XYZ | D3DFVF_NORMAL | D3DFVF_DIFFUSE | D3DFVF_TEX1) + , m_dwLeafVertexShader(D3DFVF_XYZ | D3DFVF_NORMAL | D3DFVF_DIFFUSE | D3DFVF_TEX1) { - MT_PLATFORM_STUB(); } CSpeedTreeForestDirectX8::~CSpeedTreeForestDirectX8() { - MT_PLATFORM_STUB(); + Clear(); } -auto CSpeedTreeForestDirectX8::UploadWindMatrix(unsigned int, const float *) const -> void +auto CSpeedTreeForestDirectX8::UploadWindMatrix(unsigned int uiLocation, const float * pMatrix) const -> void { - MT_PLATFORM_STUB(); + if (pMatrix) + STATEMANAGER.SetVertexShaderConstant(uiLocation, pMatrix, 4); } -auto CSpeedTreeForestDirectX8::UpdateCompundMatrix(const D3DXVECTOR3 &, const D3DXMATRIX &, const D3DXMATRIX &) -> void +auto CSpeedTreeForestDirectX8::UpdateCompundMatrix(const D3DXVECTOR3 & /*c_rEyeVec*/, const D3DXMATRIX & c_rmatView, const D3DXMATRIX & c_rmatProj) -> void { - MT_PLATFORM_STUB(); + D3DXMATRIX matBlendShader; + D3DXMatrixMultiply(&matBlendShader, &c_rmatView, &c_rmatProj); + D3DXMatrixTranspose(&matBlendShader, &matBlendShader); + STATEMANAGER.SetVertexShaderConstant(0, &matBlendShader, 4); } -auto CSpeedTreeForestDirectX8::Render(unsigned long) -> void +auto CSpeedTreeForestDirectX8::Render(unsigned long ulRenderBitVector) -> void { - MT_PLATFORM_STUB(); + if (!IsNativeTerrainRenderEnabled()) + return; + + UpdateSystem(CTimer::Instance().GetCurrentSecond()); + + if (m_pMainTreeMap.empty()) + return; + + if (!(ulRenderBitVector & Forest_RenderToShadow) && !(ulRenderBitVector & Forest_RenderToMiniMap)) + { + CCamera * pCamera = CCameraManager::Instance().GetCurrentCamera(); + if (pCamera) + UpdateCompundMatrix(pCamera->GetEye(), ms_matView, ms_matProj); + } + + DWORD dwLightState = STATEMANAGER.GetRenderState(D3DRS_LIGHTING); + DWORD dwColorVertexState = STATEMANAGER.GetRenderState(D3DRS_COLORVERTEX); + DWORD dwFogVertexMode = STATEMANAGER.GetRenderState(D3DRS_FOGVERTEXMODE); + + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, TRUE); + STATEMANAGER.SetRenderState(D3DRS_COLORVERTEX, TRUE); + + TTreeMap::iterator itor = m_pMainTreeMap.begin(); + UINT uiCount = 0; + + while (itor != m_pMainTreeMap.end()) + { + CSpeedTreeWrapper * pMainTree = (itor++)->second; + if (!pMainTree) + continue; + CSpeedTreeWrapper ** ppInstances = pMainTree->GetInstances(uiCount); + for (UINT i = 0; i < uiCount; ++i) + { + if (ppInstances[i]) + ppInstances[i]->Advance(); + } + } + + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG2, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG2, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_MINFILTER, D3DTEXF_LINEAR); + STATEMANAGER.SetTextureStageState(0, D3DTSS_MAGFILTER, D3DTEXF_LINEAR); + STATEMANAGER.SetTextureStageState(0, D3DTSS_MIPFILTER, D3DTEXF_LINEAR); + + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + + STATEMANAGER.SaveRenderState(D3DRS_ALPHATESTENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAFUNC, D3DCMP_GREATER); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAREF, 0x00000060); + STATEMANAGER.SaveRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + + STATEMANAGER.SetVertexShader(m_dwBranchVertexShader); + + // Render branches + if (ulRenderBitVector & Forest_RenderBranches) + { + STATEMANAGER.SetRenderState(D3DRS_ALPHATESTENABLE, FALSE); + itor = m_pMainTreeMap.begin(); + while (itor != m_pMainTreeMap.end()) + { + CSpeedTreeWrapper * pMainTree = (itor++)->second; + if (!pMainTree) + continue; + CSpeedTreeWrapper ** ppInstances = pMainTree->GetInstances(uiCount); + pMainTree->SetupBranchForTreeType(); + for (UINT i = 0; i < uiCount; ++i) + { + if (ppInstances[i] && ppInstances[i]->isShow()) + ppInstances[i]->RenderBranches(); + } + } + } + + // Render leaves + if (ulRenderBitVector & Forest_RenderLeaves) + { + STATEMANAGER.SetVertexShader(m_dwLeafVertexShader); + STATEMANAGER.SetRenderState(D3DRS_ALPHATESTENABLE, TRUE); + STATEMANAGER.SetRenderState(D3DRS_ALPHAFUNC, D3DCMP_GREATER); + STATEMANAGER.SetRenderState(D3DRS_ALPHAREF, 0x00000060); + STATEMANAGER.SetRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + + itor = m_pMainTreeMap.begin(); + while (itor != m_pMainTreeMap.end()) + { + CSpeedTreeWrapper * pMainTree = (itor++)->second; + if (!pMainTree) + continue; + CSpeedTreeWrapper ** ppInstances = pMainTree->GetInstances(uiCount); + pMainTree->SetupLeafForTreeType(); + for (UINT i = 0; i < uiCount; ++i) + { + if (ppInstances[i] && ppInstances[i]->isShow()) + ppInstances[i]->RenderLeaves(); + } + pMainTree->EndLeafForTreeType(); + } + } + + STATEMANAGER.SetRenderState(D3DRS_LIGHTING, dwLightState); + STATEMANAGER.SetRenderState(D3DRS_COLORVERTEX, dwColorVertexState); + STATEMANAGER.SetRenderState(D3DRS_FOGVERTEXMODE, dwFogVertexMode); + + STATEMANAGER.RestoreRenderState(D3DRS_ALPHATESTENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAFUNC); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAREF); + STATEMANAGER.RestoreRenderState(D3DRS_CULLMODE); } -auto CSpeedTreeForestDirectX8::SetRenderingDevice(LPDIRECT3DDEVICE8) -> bool +auto CSpeedTreeForestDirectX8::SetRenderingDevice(LPDIRECT3DDEVICE8 pDevice) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + m_pDx = pDevice; + return InitVertexShaders(); } auto CSpeedTreeForestDirectX8::InitVertexShaders() -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + m_dwBranchVertexShader = D3DFVF_XYZ | D3DFVF_NORMAL | D3DFVF_DIFFUSE | D3DFVF_TEX1; + m_dwLeafVertexShader = D3DFVF_XYZ | D3DFVF_NORMAL | D3DFVF_DIFFUSE | D3DFVF_TEX1; + CSpeedTreeWrapper::SetVertexShaders(m_dwBranchVertexShader, m_dwLeafVertexShader); + return true; } diff --git a/extension/src/platform/SpeedTreeLib/SpeedTreeWrapper.cpp b/extension/src/platform/SpeedTreeLib/SpeedTreeWrapper.cpp index 8791a373..cfe7c7a8 100644 --- a/extension/src/platform/SpeedTreeLib/SpeedTreeWrapper.cpp +++ b/extension/src/platform/SpeedTreeLib/SpeedTreeWrapper.cpp @@ -1,195 +1,1097 @@ -// Platform skeleton for SpeedTreeLib/SpeedTreeWrapper.h (40250 SpeedTreeLib/SpeedTreeWrapper.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. #include "SpeedTreeLib/StdAfx.h" #include "SpeedTreeLib/SpeedTreeWrapper.h" +#include "SpeedTreeLib/SpeedTreeForestDirectX8.h" -#include "../PlatformStub.h" +#include "EterBase/Filename.h" +#include "EterBase/Timer.h" +#include "EterLib/Camera.h" +#include "EterLib/ResourceManager.h" +#include "EterLib/StateManager.h" +#include "EterPack/EterPackManager.h" +#include "../EterLib/RenderCommands3D.h" -auto CSpeedTreeWrapper::OnUpdateCollisionData(const CStaticCollisionDataVector *) -> void +#include + +#include +#include +#include +#include +#include +#include +#include + +struct CSpeedTreeRT::SGeometry { - MT_PLATFORM_STUB(); + std::string species; + std::string composite_name; + float height = 1000.0f; + float variance = 0.0f; + float radius = 350.0f; + float trunk_radius = 35.0f; + float trunk_height = 750.0f; + uint32_t branch_index_count = 0; + uint32_t leaf_vertex_count = 0; +}; + +struct CSpeedTreeRT::STextures +{ + std::string branch_tex; + std::string composite_tex; + std::string shadow_tex; +}; + +float CSpeedTreeRT::GetLeafLightingAdjustment() const +{ + return 0.5f; } -auto CSpeedTreeWrapper::GetBoundingSphere(D3DXVECTOR3 &, float &) -> bool +float CSpeedTreeRT::SetWindStrength(float fNewStrength, float /*fOldStrength*/, float /*fFrequencyTimeOffset*/) { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + return fNewStrength; } -decltype(CSpeedTreeWrapper::ms_bSelfShadowOn) CSpeedTreeWrapper::ms_bSelfShadowOn{}; - -auto CSpeedTreeWrapper::SetPosition(float, float, float) -> void +void CSpeedTreeRT::GetCollisionObject(unsigned int /*nIndex*/, ECollisionObjectType & eType, float * pPosition, float * pDimensions) { - MT_PLATFORM_STUB(); + eType = CO_CYLINDER; + if (pPosition) + { + pPosition[0] = 0.0f; + pPosition[1] = 0.0f; + pPosition[2] = 0.0f; + } + if (pDimensions) + { + pDimensions[0] = 35.0f; + pDimensions[1] = 600.0f; + pDimensions[2] = 0.0f; + } } -auto CSpeedTreeWrapper::CalculateBBox() -> void +namespace { - MT_PLATFORM_STUB(); + +constexpr float kPI = 3.14159265358979323846f; +constexpr float kTAU = 2.0f * kPI; +constexpr DWORD kTreeFVF = D3DFVF_XYZ | D3DFVF_NORMAL | D3DFVF_DIFFUSE | D3DFVF_TEX1; + +std::string to_lower_str(std::string s) +{ + std::transform(s.begin(), s.end(), s.begin(), + [](unsigned char c) { return static_cast(std::tolower(c)); }); + return s; } -auto CSpeedTreeWrapper::OnRender() -> void +std::string basename_only(std::string path) { - MT_PLATFORM_STUB(); + for (char & c : path) + { + if (c == '\\') + c = '/'; + } + const size_t slash = path.find_last_of('/'); + return slash == std::string::npos ? path : path.substr(slash + 1); } -auto CSpeedTreeWrapper::OnRenderPCBlocker() -> void +std::string replace_ext_dds(std::string path) { - MT_PLATFORM_STUB(); + const size_t dot = path.find_last_of('.'); + if (dot != std::string::npos) + path.resize(dot); + return path + ".dds"; } +std::string sibling_dds_path(const std::string & spt_file, const std::string & tex_ref) +{ + if (tex_ref.empty()) + return ""; + std::string spt_copy = spt_file; + std::string dir = CFileNameHelper::GetPath(spt_copy); + std::string name = replace_ext_dds(basename_only(tex_ref)); + return dir + name; +} + +bool is_conifer_species(const std::string & hint) +{ + const std::string h = to_lower_str(hint); + static const char * kw[] = { + "cedar", "cypress", "pine", "fir", "spruce", "conifer", "juniper", "christmastree" + }; + for (const char * k : kw) + { + if (h.find(k) != std::string::npos) + return true; + } + return false; +} + +bool is_palm_species(const std::string & hint) +{ + const std::string h = to_lower_str(hint); + static const char * kw[] = { "palm", "banana", "aloe", "fern", "joshua" }; + for (const char * k : kw) + { + if (h.find(k) != std::string::npos) + return true; + } + return false; +} + +uint32_t hash32(uint32_t s) +{ + s ^= s >> 16; + s *= 0x7feb352dU; + s ^= s >> 15; + s *= 0x846ca68bU; + s ^= s >> 16; + return s; +} + +float hash01(uint32_t s) +{ + return float(hash32(s) & 0x00FFFFFFU) / float(0x01000000U); +} + +uint32_t species_seed(const std::string & s) +{ + uint32_t h = 2166136261U; + for (unsigned char c : s) + { + h ^= c; + h *= 16777619U; + } + return h; +} + +struct UVRect +{ + float u0 = 0.0f, v0 = 0.0f, u1 = 1.0f, v1 = 1.0f; +}; + +std::vector foliage_rects(const std::string & species, const std::string & composite, bool atlas) +{ + if (!atlas) + return { { 0.0f, 0.0f, 1.0f, 1.0f } }; + const std::string s = to_lower_str(species); + const std::string c = to_lower_str(composite); + const bool fall = s.find("fall") != std::string::npos; + const bool winter = s.find("winter") != std::string::npos; + if (c.find("b1") != std::string::npos) + { + if (fall) + return { { 0.00f, 0.05f, 0.25f, 0.25f }, { 0.25f, 0.25f, 0.50f, 0.50f } }; + return { + { 0.25f, 0.02f, 0.50f, 0.23f }, + { 0.25f, 0.18f, 0.50f, 0.36f }, + { 0.00f, 0.27f, 0.27f, 0.49f } + }; + } + if (c.find("b2") != std::string::npos) + { + if (fall) + return { { 0.00f, 0.00f, 0.25f, 0.25f }, { 0.25f, 0.25f, 0.50f, 0.50f } }; + return { + { 0.50f, 0.38f, 0.75f, 0.63f }, + { 0.50f, 0.63f, 0.75f, 0.88f }, + { 0.00f, 0.38f, 0.25f, 0.62f } + }; + } + if (c.find("b3") != std::string::npos) + { + if (fall) + return { { 0.25f, 0.25f, 0.50f, 0.50f } }; + return { + { 0.00f, 0.25f, 0.25f, 0.50f }, + { 0.00f, 0.50f, 0.25f, 0.75f }, + { 0.25f, 0.50f, 0.50f, 0.75f } + }; + } + if (c.find("n1") != std::string::npos) + { + if (winter) + return { { 0.00f, 0.36f, 0.50f, 0.58f }, { 0.25f, 0.55f, 0.52f, 0.75f } }; + return { { 0.00f, 0.72f, 0.28f, 0.96f }, { 0.25f, 0.74f, 0.53f, 0.97f } }; + } + if (c.find("n2") != std::string::npos) + { + return { + { 0.00f, 0.48f, 0.27f, 0.75f }, + { 0.26f, 0.73f, 0.58f, 1.00f }, + { 0.75f, 0.48f, 1.00f, 0.80f } + }; + } + return { { 0.0f, 0.0f, 1.0f, 1.0f } }; +} + +struct TreeVertex +{ + TPosition position; + TNormal normal; + DWORD diffuse; + TTextureCoordinate texCoord; +}; + +struct TreeMeshBuilder +{ + std::vector verts; + std::vector indices; + DWORD color = 0xFFFFFFFFu; + + static TreeVertex make_vtx(const D3DXVECTOR3 & p, const D3DXVECTOR3 & n, DWORD col, float u, float v) + { + TreeVertex out{}; + out.position = TPosition(p.x, p.y, p.z); + out.normal = TNormal(n.x, n.y, n.z); + out.diffuse = col; + out.texCoord = TTextureCoordinate(u, v); + return out; + } + + void add_tube_quad( + const D3DXVECTOR3 & b0, + const D3DXVECTOR3 & b1, + const D3DXVECTOR3 & t1, + const D3DXVECTOR3 & t0, + const D3DXVECTOR3 & n0, + const D3DXVECTOR3 & n1, + float u0, + float u1, + float v0, + float v1) + { + if (verts.size() + 4 > 65530) + return; + const uint16_t base = static_cast(verts.size()); + verts.push_back(make_vtx(b0, n0, color, u0, v0)); + verts.push_back(make_vtx(b1, n1, color, u1, v0)); + verts.push_back(make_vtx(t1, n1, color, u1, v1)); + verts.push_back(make_vtx(t0, n0, color, u0, v1)); + indices.push_back(base + 0); + indices.push_back(base + 2); + indices.push_back(base + 1); + indices.push_back(base + 0); + indices.push_back(base + 3); + indices.push_back(base + 2); + } + + void add_leaf_quad_unindexed( + const D3DXVECTOR3 & a, + const D3DXVECTOR3 & b, + const D3DXVECTOR3 & c, + const D3DXVECTOR3 & d, + const UVRect & r, + bool flip_u) + { + const D3DXVECTOR3 ab = b - a; + const D3DXVECTOR3 ad = d - a; + D3DXVECTOR3 nn; + D3DXVec3Cross(&nn, &ab, &ad); + D3DXVec3Normalize(&nn, &nn); + // Bias leaf normals upward slightly for softer canopy lighting + nn.z = std::fabs(nn.z) * 0.5f + 0.5f; + D3DXVec3Normalize(&nn, &nn); + + const float l = flip_u ? r.u1 : r.u0; + const float rr = flip_u ? r.u0 : r.u1; + const TreeVertex v0 = make_vtx(a, nn, color, l, r.v1); + const TreeVertex v1 = make_vtx(b, nn, color, rr, r.v1); + const TreeVertex v2 = make_vtx(c, nn, color, rr, r.v0); + const TreeVertex v3 = make_vtx(d, nn, color, l, r.v0); + + verts.push_back(v0); + verts.push_back(v1); + verts.push_back(v2); + verts.push_back(v0); + verts.push_back(v2); + verts.push_back(v3); + } +}; + +// Z-up tube generator (from/to in Metin2 Z-up centimeters) +void add_tube_zup( + TreeMeshBuilder & m, + const D3DXVECTOR3 & from, + const D3DXVECTOR3 & to, + float r0, + float r1, + int seg, + float bark_repeat) +{ + D3DXVECTOR3 axis = to - from; + const float len = D3DXVec3Length(&axis); + if (len < 1e-3f) + return; + axis /= len; + + const D3DXVECTOR3 helper = (std::fabs(axis.z) > 0.9f) ? D3DXVECTOR3(1.0f, 0.0f, 0.0f) : D3DXVECTOR3(0.0f, 0.0f, 1.0f); + D3DXVECTOR3 u, w; + D3DXVec3Cross(&u, &axis, &helper); + D3DXVec3Normalize(&u, &u); + D3DXVec3Cross(&w, &axis, &u); + D3DXVec3Normalize(&w, &w); + + for (int i = 0; i < seg; ++i) + { + const float a0 = float(i) / float(seg) * kTAU; + const float a1 = float(i + 1) / float(seg) * kTAU; + const D3DXVECTOR3 n0 = u * std::cos(a0) + w * std::sin(a0); + const D3DXVECTOR3 n1 = u * std::cos(a1) + w * std::sin(a1); + m.add_tube_quad( + from + n0 * r0, + from + n1 * r0, + to + n1 * r1, + to + n0 * r1, + n0, + n1, + float(i) / float(seg), + float(i + 1) / float(seg), + bark_repeat, + 0.0f); + } +} + +void build_branches_zup( + TreeMeshBuilder & wood, + const std::string & species, + float H, + bool conifer, + bool palm) +{ + const float trunk_top = H * (palm ? 0.82f : (conifer ? 0.90f : 0.76f)); + const float trunk_r = H * (palm ? 0.028f : 0.035f); + const float H_m = H * 0.01f; + add_tube_zup( + wood, + D3DXVECTOR3(0.0f, 0.0f, 0.0f), + D3DXVECTOR3(0.0f, 0.0f, trunk_top), + trunk_r * 1.35f, + trunk_r * 0.42f, + 9, + H_m * 0.22f); + if (palm) + return; + + const int count = conifer ? 9 : 8; + const uint32_t seed = species_seed(species); + for (int i = 0; i < count; ++i) + { + const float f = (i + 1.0f) / (count + 1.0f); + const float z = H * (conifer ? (0.28f + f * 0.52f) : (0.32f + f * 0.34f)); + const float angle = kTAU * (f * 1.6180339f + hash01(seed + i * 17U)); + const float len = H * (conifer + ? (0.24f * (1.0f - f * 0.55f)) + : (0.18f + 0.08f * hash01(seed + i * 29U))); + const D3DXVECTOR3 from(0.0f, 0.0f, z); + const D3DXVECTOR3 to( + std::cos(angle) * len, + std::sin(angle) * len, + z + H * (conifer ? 0.06f : (0.10f + 0.06f * hash01(seed + i * 31U)))); + add_tube_zup( + wood, + from, + to, + trunk_r * (0.55f - 0.20f * f), + trunk_r * 0.12f, + 6, + H_m * 0.08f); + if (!conifer && (i % 2 == 0)) + { + const float side = angle + (hash01(seed + i * 37U) > 0.5f ? 0.65f : -0.65f); + const D3DXVECTOR3 tip = to + D3DXVECTOR3(std::cos(side), std::sin(side), 0.65f) * (len * 0.42f); + add_tube_zup(wood, to, tip, trunk_r * 0.16f, trunk_r * 0.05f, 5, H_m * 0.04f); + } + } +} + +void build_leaves_zup( + TreeMeshBuilder & leaves, + const std::string & species, + float H, + bool conifer, + bool palm, + const std::vector & rects) +{ + const uint32_t seed = species_seed(species); + const int count = palm ? 16 : 24; + for (int i = 0; i < count; ++i) + { + const float a = kTAU * (float(i) * 0.6180339f + hash01(seed + i * 101U) * 0.15f); + D3DXVECTOR3 center; + float width = 1.0f; + float height = 1.0f; + if (palm) + { + const float radial = H * (0.10f + 0.18f * hash01(seed + i * 103U)); + center = D3DXVECTOR3( + std::cos(a) * radial, + std::sin(a) * radial, + H * (0.78f + 0.12f * hash01(seed + i * 107U))); + width = H * 0.32f; + height = H * 0.18f; + } + else if (conifer) + { + const float zf = 0.30f + 0.62f * (float(i) + 0.5f) / float(count); + const float radial = H * 0.23f * (1.0f - zf * 0.70f) * + (0.35f + 0.65f * hash01(seed + i * 109U)); + center = D3DXVECTOR3(std::cos(a) * radial, std::sin(a) * radial, H * zf); + width = H * (0.18f + 0.10f * (1.0f - zf)); + height = H * 0.18f; + } + else + { + const float zf = hash01(seed + i * 109U); + const float zn = zf * 2.0f - 1.0f; + const float radial = H * 0.34f * std::sqrt(std::max(0.05f, 1.0f - zn * zn)) * + (0.25f + 0.75f * std::sqrt(hash01(seed + i * 113U))); + center = D3DXVECTOR3( + std::cos(a) * radial, + std::sin(a) * radial, + H * (0.58f + zf * 0.34f)); + width = H * (0.23f + 0.10f * hash01(seed + i * 127U)); + height = H * (0.15f + 0.08f * hash01(seed + i * 131U)); + } + + const D3DXVECTOR3 right(std::cos(a + kPI * 0.5f), std::sin(a + kPI * 0.5f), 0.0f); + const D3DXVECTOR3 up(0.0f, 0.0f, 1.0f); + const UVRect & uv = rects[size_t(i) % rects.size()]; + + auto add_card = [&](const D3DXVECTOR3 & r, bool flip) { + leaves.add_leaf_quad_unindexed( + center - r * (width * 0.5f) - up * (height * 0.5f), + center + r * (width * 0.5f) - up * (height * 0.5f), + center + r * (width * 0.5f) + up * (height * 0.5f), + center - r * (width * 0.5f) + up * (height * 0.5f), + uv, + flip); + }; + add_card(right, (i & 1) != 0); + const D3DXVECTOR3 crossed(std::cos(a), std::sin(a), 0.0f); + add_card(crossed, (i & 1) == 0); + } +} + +} // namespace + +decltype(CSpeedTreeWrapper::ms_bSelfShadowOn) CSpeedTreeWrapper::ms_bSelfShadowOn = true; +decltype(CSpeedTreeWrapper::ms_dwBranchVertexShader) CSpeedTreeWrapper::ms_dwBranchVertexShader = kTreeFVF; +decltype(CSpeedTreeWrapper::ms_dwLeafVertexShader) CSpeedTreeWrapper::ms_dwLeafVertexShader = kTreeFVF; + CSpeedTreeWrapper::CSpeedTreeWrapper() + : m_pSpeedTree(new CSpeedTreeRT) + , m_pTextureInfo(NULL) + , m_bIsInstance(false) + , m_pInstanceOf(NULL) + , m_pGeometryCache(NULL) + , m_pBranchVertexBuffer(NULL) + , m_unBranchVertexCount(0) + , m_pBranchIndexBuffer(NULL) + , m_pBranchIndexCounts(NULL) + , m_pFrondVertexBuffer(NULL) + , m_unFrondVertexCount(0) + , m_pFrondIndexBuffer(NULL) + , m_pFrondIndexCounts(NULL) + , m_usNumLeafLods(0) + , m_pLeafVertexBuffer(NULL) + , m_pLeavesUpdatedByCpu(NULL) { - MT_PLATFORM_STUB(); + m_afPos[0] = m_afPos[1] = m_afPos[2] = 0.0f; + std::memset(m_afBoundingBox, 0, sizeof(m_afBoundingBox)); } CSpeedTreeWrapper::~CSpeedTreeWrapper() { - MT_PLATFORM_STUB(); + if (!m_bIsInstance) + { + SAFE_RELEASE(m_pBranchVertexBuffer); + SAFE_RELEASE(m_pBranchIndexBuffer); + SAFE_DELETE_ARRAY(m_pBranchIndexCounts); + + SAFE_RELEASE(m_pFrondVertexBuffer); + SAFE_RELEASE(m_pFrondIndexBuffer); + SAFE_DELETE_ARRAY(m_pFrondIndexCounts); + + if (m_pLeafVertexBuffer) + { + for (unsigned short i = 0; i < m_usNumLeafLods; ++i) + SAFE_RELEASE(m_pLeafVertexBuffer[i]); + SAFE_DELETE_ARRAY(m_pLeafVertexBuffer); + } + SAFE_DELETE_ARRAY(m_pLeavesUpdatedByCpu); + SAFE_DELETE(m_pTextureInfo); + SAFE_DELETE(m_pGeometryCache); + } + + SAFE_DELETE(m_pSpeedTree); + Clear(); +} + +auto CSpeedTreeWrapper::SetVertexShaders(DWORD dwBranchVertexShader, DWORD dwLeafVertexShader) -> void +{ + ms_dwBranchVertexShader = dwBranchVertexShader; + ms_dwLeafVertexShader = dwLeafVertexShader; } auto CSpeedTreeWrapper::GetPosition() -> const float * { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + return m_afPos; } -auto CSpeedTreeWrapper::SetVertexShaders(DWORD, DWORD) -> void +auto CSpeedTreeWrapper::SetPosition(float x, float y, float z) -> void { - MT_PLATFORM_STUB(); + m_afPos[0] = x; + m_afPos[1] = y; + m_afPos[2] = z; + CGraphicObjectInstance::SetPosition(x, y, z); + CGraphicObjectInstance::Transform(); + CalculateBBox(); } -auto CSpeedTreeWrapper::LoadTree(const char *, const BYTE *, unsigned int, unsigned int, float, float) -> bool +auto CSpeedTreeWrapper::GetBoundingSphere(D3DXVECTOR3 & v3Center, float & fRadius) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + const float fX = m_afBoundingBox[3] - m_afBoundingBox[0]; + const float fY = m_afBoundingBox[4] - m_afBoundingBox[1]; + const float fZ = m_afBoundingBox[5] - m_afBoundingBox[2]; + + v3Center.x = m_afPos[0]; + v3Center.y = m_afPos[1]; + v3Center.z = m_afPos[2] + fZ * 0.5f; + + fRadius = std::sqrt(fX * fX + fY * fY + fZ * fZ) * 0.5f * 0.9f; + return true; } -auto CSpeedTreeWrapper::GetTreeSize(float &, float &) -> void +auto CSpeedTreeWrapper::CalculateBBox() -> void { - MT_PLATFORM_STUB(); + const float fX = m_afBoundingBox[3] - m_afBoundingBox[0]; + const float fY = m_afBoundingBox[4] - m_afBoundingBox[1]; + const float fZ = m_afBoundingBox[5] - m_afBoundingBox[2]; + + m_v3BBoxMin = D3DXVECTOR3(-fX * 0.5f, -fY * 0.5f, 0.0f); + m_v3BBoxMax = D3DXVECTOR3(+fX * 0.5f, +fY * 0.5f, fZ); + + m_v4TBBox[0] = D3DXVECTOR4(m_v3BBoxMin.x, m_v3BBoxMin.y, m_v3BBoxMin.z, 1.0f); + m_v4TBBox[1] = D3DXVECTOR4(m_v3BBoxMin.x, m_v3BBoxMax.y, m_v3BBoxMin.z, 1.0f); + m_v4TBBox[2] = D3DXVECTOR4(m_v3BBoxMax.x, m_v3BBoxMin.y, m_v3BBoxMin.z, 1.0f); + m_v4TBBox[3] = D3DXVECTOR4(m_v3BBoxMax.x, m_v3BBoxMax.y, m_v3BBoxMin.z, 1.0f); + m_v4TBBox[4] = D3DXVECTOR4(m_v3BBoxMin.x, m_v3BBoxMin.y, m_v3BBoxMax.z, 1.0f); + m_v4TBBox[5] = D3DXVECTOR4(m_v3BBoxMin.x, m_v3BBoxMax.y, m_v3BBoxMax.z, 1.0f); + m_v4TBBox[6] = D3DXVECTOR4(m_v3BBoxMax.x, m_v3BBoxMin.y, m_v3BBoxMax.z, 1.0f); + m_v4TBBox[7] = D3DXVECTOR4(m_v3BBoxMax.x, m_v3BBoxMax.y, m_v3BBoxMax.z, 1.0f); + + const D3DXMATRIX & c_rmatTransform = GetTransform(); + for (DWORD i = 0; i < 8; ++i) + { + D3DXVec4Transform(&m_v4TBBox[i], &m_v4TBBox[i], &c_rmatTransform); + if (0 == i) + { + m_v3TBBoxMin = D3DXVECTOR3(m_v4TBBox[i].x, m_v4TBBox[i].y, m_v4TBBox[i].z); + m_v3TBBoxMax = m_v3TBBoxMin; + } + else + { + m_v3TBBoxMin.x = std::min(m_v3TBBoxMin.x, m_v4TBBox[i].x); + m_v3TBBoxMax.x = std::max(m_v3TBBoxMax.x, m_v4TBBox[i].x); + m_v3TBBoxMin.y = std::min(m_v3TBBoxMin.y, m_v4TBBox[i].y); + m_v3TBBoxMax.y = std::max(m_v3TBBoxMax.y, m_v4TBBox[i].y); + m_v3TBBoxMin.z = std::min(m_v3TBBoxMin.z, m_v4TBBox[i].z); + m_v3TBBoxMax.z = std::max(m_v3TBBoxMax.z, m_v4TBBox[i].z); + } + } +} + +auto CSpeedTreeWrapper::OnUpdateCollisionData(const CStaticCollisionDataVector * /*pscdVector*/) -> void +{ + ClearCollision(); + D3DXMATRIX mat; + D3DXMatrixTranslation(&mat, m_afPos[0], m_afPos[1], m_afPos[2]); + + for (UINT i = 0; i < GetCollisionObjectCount(); ++i) + { + CSpeedTreeRT::ECollisionObjectType ObjectType = CSpeedTreeRT::CO_CYLINDER; + CStaticCollisionData CollisionData{}; + GetCollisionObject(i, ObjectType, reinterpret_cast(&CollisionData.v3Position), CollisionData.fDimensions); + if (ObjectType == CSpeedTreeRT::CO_BOX) + continue; + if (ObjectType == CSpeedTreeRT::CO_SPHERE) + CollisionData.dwType = COLLISION_TYPE_SPHERE; + else if (ObjectType == CSpeedTreeRT::CO_CYLINDER) + CollisionData.dwType = COLLISION_TYPE_CYLINDER; + AddCollision(&CollisionData, &mat); + } } auto CSpeedTreeWrapper::GetCollisionObjectCount() -> UINT { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + return 1; } -auto CSpeedTreeWrapper::GetCollisionObject(unsigned int, CSpeedTreeRT::ECollisionObjectType &, float *, float *) -> void +auto CSpeedTreeWrapper::GetCollisionObject( + unsigned int /*nIndex*/, + CSpeedTreeRT::ECollisionObjectType & eType, + float * pPosition, + float * pDimensions) -> void { - MT_PLATFORM_STUB(); + eType = CSpeedTreeRT::CO_CYLINDER; + if (pPosition) + { + pPosition[0] = 0.0f; + pPosition[1] = 0.0f; + pPosition[2] = 0.0f; + } + if (pDimensions) + { + pDimensions[0] = m_pGeometryCache ? m_pGeometryCache->trunk_radius : 35.0f; + pDimensions[1] = m_pGeometryCache ? m_pGeometryCache->trunk_height : 600.0f; + pDimensions[2] = 0.0f; + } } -auto CSpeedTreeWrapper::SetupBranchForTreeType() const -> void +auto CSpeedTreeWrapper::GetTreeSize(float & r_fSize, float & r_fVariance) -> void { - MT_PLATFORM_STUB(); + r_fSize = m_pGeometryCache ? m_pGeometryCache->height : 1000.0f; + r_fVariance = m_pGeometryCache ? m_pGeometryCache->variance : 0.0f; } -auto CSpeedTreeWrapper::SetupFrondForTreeType() const -> void +auto CSpeedTreeWrapper::LoadTree( + const char * pszSptFile, + const BYTE * c_pbBlock, + unsigned int uiBlockSize, + unsigned int /*nSeed*/, + float fSize, + float fSizeVariance) -> bool { - MT_PLATFORM_STUB(); -} + if (!IsNativeTerrainRenderEnabled()) + return false; -auto CSpeedTreeWrapper::SetupLeafForTreeType() const -> void -{ - MT_PLATFORM_STUB(); -} + const std::string spt_path = pszSptFile ? pszSptFile : ""; + std::string bytes; + if (c_pbBlock && uiBlockSize > 0) + { + bytes.assign(reinterpret_cast(c_pbBlock), uiBlockSize); + } + else if (!spt_path.empty()) + { + CMappedFile file; + LPCVOID pvData = NULL; + if (CEterPackManager::Instance().Get(file, spt_path.c_str(), &pvData) && pvData && file.Size() > 0) + bytes.assign(static_cast(pvData), file.Size()); + } -auto CSpeedTreeWrapper::EndLeafForTreeType() -> void -{ - MT_PLATFORM_STUB(); -} + fmt::SptInfo info; + if (!bytes.empty()) + fmt::sniff_spt(bytes, info); -auto CSpeedTreeWrapper::RenderBranches() const -> void -{ - MT_PLATFORM_STUB(); -} + m_pTextureInfo = new CSpeedTreeRT::STextures; + for (const std::string & ref : info.texture_refs) + { + if (to_lower_str(ref).find("bark") != std::string::npos) + { + m_pTextureInfo->branch_tex = ref; + break; + } + } + if (m_pTextureInfo->branch_tex.empty() && !info.texture_refs.empty()) + m_pTextureInfo->branch_tex = info.texture_refs.front(); + m_pTextureInfo->composite_tex = info.composite_texture; + m_pTextureInfo->shadow_tex = info.self_shadow_texture; -auto CSpeedTreeWrapper::RenderFronds() const -> void -{ - MT_PLATFORM_STUB(); -} + if (!m_pTextureInfo->branch_tex.empty()) + { + const std::string p = sibling_dds_path(spt_path, m_pTextureInfo->branch_tex); + LoadTexture(p.c_str(), m_BranchImageInstance); + } + if (!m_pTextureInfo->composite_tex.empty()) + { + const std::string p = sibling_dds_path(spt_path, m_pTextureInfo->composite_tex); + LoadTexture(p.c_str(), m_CompositeImageInstance); + } + if (!m_pTextureInfo->shadow_tex.empty()) + { + const std::string p = sibling_dds_path(spt_path, m_pTextureInfo->shadow_tex); + LoadTexture(p.c_str(), m_ShadowImageInstance); + } -auto CSpeedTreeWrapper::RenderLeaves() const -> void -{ - MT_PLATFORM_STUB(); -} + const float H = (fSize > 150.0f) ? fSize : 1000.0f; + const float R = H * 0.35f; -auto CSpeedTreeWrapper::RenderBillboards() const -> void -{ - MT_PLATFORM_STUB(); -} + m_afBoundingBox[0] = -R; + m_afBoundingBox[1] = -R; + m_afBoundingBox[2] = 0.0f; + m_afBoundingBox[3] = +R; + m_afBoundingBox[4] = +R; + m_afBoundingBox[5] = H; -auto CSpeedTreeWrapper::GetInstances(unsigned int &) -> CSpeedTreeWrapper ** -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} + m_pGeometryCache = new CSpeedTreeRT::SGeometry; + m_pGeometryCache->species = spt_path; + m_pGeometryCache->composite_name = info.composite_texture; + m_pGeometryCache->height = H; + m_pGeometryCache->variance = (fSizeVariance >= 0.0f) ? fSizeVariance : 0.0f; + m_pGeometryCache->radius = R; + m_pGeometryCache->trunk_radius = H * 0.035f; + m_pGeometryCache->trunk_height = H * 0.75f; -auto CSpeedTreeWrapper::MakeInstance() -> CSpeedTreeWrapper * -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto CSpeedTreeWrapper::DeleteInstance(CSpeedTreeWrapper *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto CSpeedTreeWrapper::Advance() -> void -{ - MT_PLATFORM_STUB(); -} - -auto CSpeedTreeWrapper::GetBranchTexture() const -> LPDIRECT3DTEXTURE8 -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto CSpeedTreeWrapper::CleanUpMemory() -> void -{ - MT_PLATFORM_STUB(); + SetupBuffers(); + return true; } auto CSpeedTreeWrapper::SetupBuffers() -> void { - MT_PLATFORM_STUB(); + SetupBranchBuffers(); + SetupFrondBuffers(); + SetupLeafBuffers(); } auto CSpeedTreeWrapper::SetupBranchBuffers() -> void { - MT_PLATFORM_STUB(); + if (!m_pGeometryCache || !ms_lpd3dDevice) + return; + + const float H = m_pGeometryCache->height; + const bool conifer = is_conifer_species(m_pGeometryCache->species); + const bool palm = is_palm_species(m_pGeometryCache->species); + + TreeMeshBuilder wood; + wood.color = m_BranchImageInstance.IsEmpty() ? 0xFF4D3621u : 0xFFFFFFFFu; + build_branches_zup(wood, m_pGeometryCache->species, H, conifer, palm); + + if (wood.verts.empty() || wood.indices.empty()) + return; + + m_unBranchVertexCount = static_cast(wood.verts.size()); + if (FAILED(ms_lpd3dDevice->CreateVertexBuffer( + m_unBranchVertexCount * sizeof(TreeVertex), + D3DUSAGE_WRITEONLY, + kTreeFVF, + D3DPOOL_MANAGED, + &m_pBranchVertexBuffer))) + { + m_unBranchVertexCount = 0; + return; + } + + BYTE * pVB = NULL; + if (SUCCEEDED(m_pBranchVertexBuffer->Lock(0, 0, &pVB, 0))) + { + std::memcpy(pVB, wood.verts.data(), wood.verts.size() * sizeof(TreeVertex)); + m_pBranchVertexBuffer->Unlock(); + } + + m_pBranchIndexCounts = new unsigned short[1]; + m_pBranchIndexCounts[0] = static_cast(wood.indices.size()); + m_pGeometryCache->branch_index_count = static_cast(wood.indices.size()); + + if (FAILED(ms_lpd3dDevice->CreateIndexBuffer( + m_pBranchIndexCounts[0] * sizeof(uint16_t), + D3DUSAGE_WRITEONLY, + D3DFMT_INDEX16, + D3DPOOL_MANAGED, + &m_pBranchIndexBuffer))) + { + m_pBranchIndexCounts[0] = 0; + return; + } + + BYTE * pIB = NULL; + if (SUCCEEDED(m_pBranchIndexBuffer->Lock(0, 0, &pIB, 0))) + { + std::memcpy(pIB, wood.indices.data(), wood.indices.size() * sizeof(uint16_t)); + m_pBranchIndexBuffer->Unlock(); + } } auto CSpeedTreeWrapper::SetupFrondBuffers() -> void { - MT_PLATFORM_STUB(); + m_unFrondVertexCount = 0; } auto CSpeedTreeWrapper::SetupLeafBuffers() -> void { - MT_PLATFORM_STUB(); + if (!m_pGeometryCache || !ms_lpd3dDevice) + return; + + const float H = m_pGeometryCache->height; + const bool conifer = is_conifer_species(m_pGeometryCache->species); + const bool palm = is_palm_species(m_pGeometryCache->species); + const bool has_atlas = !m_CompositeImageInstance.IsEmpty(); + const std::vector rects = foliage_rects( + m_pGeometryCache->species, + m_pGeometryCache->composite_name, + has_atlas); + + TreeMeshBuilder leaves; + leaves.color = has_atlas ? 0xFFFFFFFFu : (conifer ? 0xFF1F471Cu : 0xFF2E6624u); + build_leaves_zup(leaves, m_pGeometryCache->species, H, conifer, palm, rects); + + if (leaves.verts.empty()) + return; + + m_usNumLeafLods = 1; + m_pLeafVertexBuffer = new LPDIRECT3DVERTEXBUFFER8[1]; + m_pLeafVertexBuffer[0] = NULL; + m_pLeavesUpdatedByCpu = new bool[1]; + m_pLeavesUpdatedByCpu[0] = true; + + m_pGeometryCache->leaf_vertex_count = static_cast(leaves.verts.size()); + + if (FAILED(ms_lpd3dDevice->CreateVertexBuffer( + leaves.verts.size() * sizeof(TreeVertex), + D3DUSAGE_WRITEONLY, + kTreeFVF, + D3DPOOL_MANAGED, + &m_pLeafVertexBuffer[0]))) + { + m_pGeometryCache->leaf_vertex_count = 0; + return; + } + + BYTE * pVB = NULL; + if (SUCCEEDED(m_pLeafVertexBuffer[0]->Lock(0, 0, &pVB, 0))) + { + std::memcpy(pVB, leaves.verts.data(), leaves.verts.size() * sizeof(TreeVertex)); + m_pLeafVertexBuffer[0]->Unlock(); + } +} + +auto CSpeedTreeWrapper::SetupBranchForTreeType() const -> void +{ + LPDIRECT3DTEXTURE8 lpd3dTexture = NULL; + if (!m_BranchImageInstance.IsEmpty()) + lpd3dTexture = const_cast(m_BranchImageInstance).GetTextureReference().GetD3DTexture(); + STATEMANAGER.SetTexture(0, lpd3dTexture); + STATEMANAGER.SetTexture(1, NULL); + + if (m_unBranchVertexCount > 0 && m_pBranchVertexBuffer && m_pBranchIndexBuffer) + { + STATEMANAGER.SetStreamSource(0, m_pBranchVertexBuffer, sizeof(TreeVertex)); + STATEMANAGER.SetIndices(m_pBranchIndexBuffer, 0); + } +} + +auto CSpeedTreeWrapper::RenderBranches() const -> void +{ + if (!m_pGeometryCache || m_unBranchVertexCount == 0 || !m_pBranchIndexCounts || m_pBranchIndexCounts[0] < 3) + return; + + PositionTree(); + STATEMANAGER.DrawIndexedPrimitive( + D3DPT_TRIANGLELIST, + 0, + m_unBranchVertexCount, + 0, + m_pBranchIndexCounts[0] / 3); +} + +auto CSpeedTreeWrapper::SetupFrondForTreeType() const -> void +{ +} + +auto CSpeedTreeWrapper::RenderFronds() const -> void +{ +} + +auto CSpeedTreeWrapper::SetupLeafForTreeType() const -> void +{ + LPDIRECT3DTEXTURE8 lpd3dTexture = NULL; + if (!m_CompositeImageInstance.IsEmpty()) + lpd3dTexture = const_cast(m_CompositeImageInstance).GetTextureReference().GetD3DTexture(); + STATEMANAGER.SetTexture(0, lpd3dTexture); + STATEMANAGER.SetTexture(1, NULL); + + if (m_pLeafVertexBuffer && m_pLeafVertexBuffer[0]) + STATEMANAGER.SetStreamSource(0, m_pLeafVertexBuffer[0], sizeof(TreeVertex)); +} + +auto CSpeedTreeWrapper::RenderLeaves() const -> void +{ + if (!m_pGeometryCache || m_pGeometryCache->leaf_vertex_count < 3 || !m_pLeafVertexBuffer || !m_pLeafVertexBuffer[0]) + return; + + PositionTree(); + STATEMANAGER.DrawPrimitive(D3DPT_TRIANGLELIST, 0, m_pGeometryCache->leaf_vertex_count / 3); +} + +auto CSpeedTreeWrapper::EndLeafForTreeType() -> void +{ +} + +auto CSpeedTreeWrapper::RenderBillboards() const -> void +{ +} + +auto CSpeedTreeWrapper::OnRender() -> void +{ + if (!IsNativeTerrainRenderEnabled()) + return; + + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG2, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG2, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + + STATEMANAGER.SaveRenderState(D3DRS_LIGHTING, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_COLORVERTEX, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHATESTENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAFUNC, D3DCMP_GREATER); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAREF, 0x00000060); + STATEMANAGER.SaveRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + + STATEMANAGER.SetVertexShader(ms_dwBranchVertexShader); + SetupBranchForTreeType(); + RenderBranches(); + + STATEMANAGER.SetVertexShader(ms_dwLeafVertexShader); + STATEMANAGER.SetRenderState(D3DRS_ALPHATESTENABLE, TRUE); + SetupLeafForTreeType(); + RenderLeaves(); + EndLeafForTreeType(); + + STATEMANAGER.RestoreRenderState(D3DRS_LIGHTING); + STATEMANAGER.RestoreRenderState(D3DRS_COLORVERTEX); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHATESTENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAFUNC); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAREF); + STATEMANAGER.RestoreRenderState(D3DRS_CULLMODE); +} + +auto CSpeedTreeWrapper::OnRenderPCBlocker() -> void +{ + if (!IsNativeTerrainRenderEnabled()) + return; + + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLORARG2, D3DTA_DIFFUSE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_COLOROP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG1, D3DTA_TEXTURE); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAARG2, D3DTA_TFACTOR); + STATEMANAGER.SetTextureStageState(0, D3DTSS_ALPHAOP, D3DTOP_MODULATE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_COLOROP, D3DTOP_DISABLE); + STATEMANAGER.SetTextureStageState(1, D3DTSS_ALPHAOP, D3DTOP_DISABLE); + + STATEMANAGER.SaveRenderState(D3DRS_LIGHTING, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_COLORVERTEX, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHABLENDENABLE, TRUE); + STATEMANAGER.SaveRenderState(D3DRS_SRCBLEND, D3DBLEND_SRCALPHA); + STATEMANAGER.SaveRenderState(D3DRS_DESTBLEND, D3DBLEND_INVSRCALPHA); + STATEMANAGER.SaveRenderState(D3DRS_TEXTUREFACTOR, 0x80FFFFFF); + STATEMANAGER.SaveRenderState(D3DRS_ALPHATESTENABLE, FALSE); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAFUNC, D3DCMP_GREATER); + STATEMANAGER.SaveRenderState(D3DRS_ALPHAREF, 0x00000030); + STATEMANAGER.SaveRenderState(D3DRS_CULLMODE, D3DCULL_NONE); + + STATEMANAGER.SetVertexShader(ms_dwBranchVertexShader); + SetupBranchForTreeType(); + RenderBranches(); + + STATEMANAGER.SetVertexShader(ms_dwLeafVertexShader); + STATEMANAGER.SetRenderState(D3DRS_ALPHATESTENABLE, TRUE); + SetupLeafForTreeType(); + RenderLeaves(); + EndLeafForTreeType(); + + STATEMANAGER.RestoreRenderState(D3DRS_LIGHTING); + STATEMANAGER.RestoreRenderState(D3DRS_COLORVERTEX); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHABLENDENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_SRCBLEND); + STATEMANAGER.RestoreRenderState(D3DRS_DESTBLEND); + STATEMANAGER.RestoreRenderState(D3DRS_TEXTUREFACTOR); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHATESTENABLE); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAFUNC); + STATEMANAGER.RestoreRenderState(D3DRS_ALPHAREF); + STATEMANAGER.RestoreRenderState(D3DRS_CULLMODE); +} + +auto CSpeedTreeWrapper::GetInstances(unsigned int & nCount) -> CSpeedTreeWrapper ** +{ + nCount = static_cast(m_vInstances.size()); + return nCount ? &m_vInstances[0] : NULL; +} + +auto CSpeedTreeWrapper::MakeInstance() -> CSpeedTreeWrapper * +{ + CSpeedTreeWrapper * pInstance = new CSpeedTreeWrapper; + pInstance->m_bIsInstance = true; + pInstance->m_cBranchMaterial = m_cBranchMaterial; + pInstance->m_cLeafMaterial = m_cLeafMaterial; + pInstance->m_cFrondMaterial = m_cFrondMaterial; + + if (!m_CompositeImageInstance.IsEmpty()) + pInstance->m_CompositeImageInstance.SetImagePointer(m_CompositeImageInstance.GetGraphicImagePointer()); + if (!m_BranchImageInstance.IsEmpty()) + pInstance->m_BranchImageInstance.SetImagePointer(m_BranchImageInstance.GetGraphicImagePointer()); + if (!m_ShadowImageInstance.IsEmpty()) + pInstance->m_ShadowImageInstance.SetImagePointer(m_ShadowImageInstance.GetGraphicImagePointer()); + + pInstance->m_pTextureInfo = m_pTextureInfo; + pInstance->m_pGeometryCache = m_pGeometryCache; + + pInstance->m_pBranchIndexBuffer = m_pBranchIndexBuffer; + pInstance->m_pBranchIndexCounts = m_pBranchIndexCounts; + pInstance->m_pBranchVertexBuffer = m_pBranchVertexBuffer; + pInstance->m_unBranchVertexCount = m_unBranchVertexCount; + + pInstance->m_pFrondIndexBuffer = m_pFrondIndexBuffer; + pInstance->m_pFrondIndexCounts = m_pFrondIndexCounts; + pInstance->m_pFrondVertexBuffer = m_pFrondVertexBuffer; + pInstance->m_unFrondVertexCount = m_unFrondVertexCount; + + pInstance->m_pLeafVertexBuffer = m_pLeafVertexBuffer; + pInstance->m_usNumLeafLods = m_usNumLeafLods; + pInstance->m_pLeavesUpdatedByCpu = m_pLeavesUpdatedByCpu; + + std::memcpy(pInstance->m_afPos, m_afPos, sizeof(m_afPos)); + std::memcpy(pInstance->m_afBoundingBox, m_afBoundingBox, sizeof(m_afBoundingBox)); + pInstance->m_pInstanceOf = this; + m_vInstances.push_back(pInstance); + return pInstance; +} + +auto CSpeedTreeWrapper::DeleteInstance(CSpeedTreeWrapper * pInstance) -> void +{ + if (!pInstance) + return; + auto it = std::find(m_vInstances.begin(), m_vInstances.end(), pInstance); + if (it != m_vInstances.end()) + m_vInstances.erase(it); + delete pInstance; +} + +auto CSpeedTreeWrapper::Advance() -> void +{ +} + +auto CSpeedTreeWrapper::GetBranchTexture() const -> LPDIRECT3DTEXTURE8 +{ + if (m_BranchImageInstance.IsEmpty()) + return NULL; + return const_cast(m_BranchImageInstance).GetTextureReference().GetD3DTexture(); +} + +auto CSpeedTreeWrapper::CleanUpMemory() -> void +{ } auto CSpeedTreeWrapper::PositionTree() const -> void { - MT_PLATFORM_STUB(); + D3DXMATRIX matTranslation; + D3DXMatrixTranslation(&matTranslation, m_afPos[0], m_afPos[1], m_afPos[2]); + STATEMANAGER.SetTransform(D3DTS_WORLD, &matTranslation); } -auto CSpeedTreeWrapper::LoadTexture(const char *, CGraphicImageInstance &) -> bool +auto CSpeedTreeWrapper::LoadTexture(const char * pFilename, CGraphicImageInstance & rImage) -> bool { - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); + if (!pFilename || !pFilename[0]) + return false; + CResource * pResource = CResourceManager::Instance().GetResourcePointer(pFilename); + if (!pResource) + return false; + rImage.SetImagePointer(static_cast(pResource)); + return !rImage.IsEmpty(); } -auto CSpeedTreeWrapper::SetShaderConstants(const float *) const -> void +auto CSpeedTreeWrapper::SetShaderConstants(const float * /*pMaterial*/) const -> void { - MT_PLATFORM_STUB(); } - -decltype(CSpeedTreeWrapper::ms_dwBranchVertexShader) CSpeedTreeWrapper::ms_dwBranchVertexShader{}; - -decltype(CSpeedTreeWrapper::ms_dwLeafVertexShader) CSpeedTreeWrapper::ms_dwLeafVertexShader{}; diff --git a/extension/src/platform/SphereLib/frustum.cpp b/extension/src/platform/SphereLib/frustum.cpp deleted file mode 100644 index 2a65351c..00000000 --- a/extension/src/platform/SphereLib/frustum.cpp +++ /dev/null @@ -1,22 +0,0 @@ -// Platform skeleton for SphereLib/frustum.h (40250 SphereLib/frustum.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. -#include "SphereLib/StdAfx.h" -#include "SphereLib/frustum.h" - -#include "../PlatformStub.h" - -auto Frustum::BuildViewFrustum(D3DXMATRIX &) -> void -{ - MT_PLATFORM_STUB(); -} - -auto Frustum::BuildViewFrustum2(D3DXMATRIX &, float, float, float, float, const D3DXVECTOR3 &, const D3DXVECTOR3 &) -> void -{ - MT_PLATFORM_STUB(); -} - -auto Frustum::ViewVolumeTest(const Vector3d &, const float) const -> ViewState -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} diff --git a/extension/src/platform/SphereLib/sphere.cpp b/extension/src/platform/SphereLib/sphere.cpp deleted file mode 100644 index 250a34d8..00000000 --- a/extension/src/platform/SphereLib/sphere.cpp +++ /dev/null @@ -1,49 +0,0 @@ -// Platform skeleton for SphereLib/sphere.h (40250 SphereLib/sphere.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. -#include "SphereLib/StdAfx.h" -#include "SphereLib/sphere.h" - -#include "../PlatformStub.h" - -SphereInterface::SphereInterface() -{ - MT_PLATFORM_STUB(); -} - -SphereInterface::~SphereInterface() -{ - MT_PLATFORM_STUB(); -} - -auto Sphere::Set(const Vector3d &, float) -> void -{ - MT_PLATFORM_STUB(); -} - -auto Sphere::Compute(const SphereInterface &) -> void -{ - MT_PLATFORM_STUB(); -} - -auto Sphere::RayIntersection(const Vector3d &, const Vector3d &, float, Vector3d *) -> bool -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto Sphere::RayIntersection(const Vector3d &, const Vector3d &, Vector3d *) -> bool -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto Sphere::RayIntersectionInFront(const Vector3d &, const Vector3d &, Vector3d *) -> bool -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto Sphere::Report() -> void -{ - MT_PLATFORM_STUB(); -} diff --git a/extension/src/platform/SphereLib/spherepack.cpp b/extension/src/platform/SphereLib/spherepack.cpp deleted file mode 100644 index 585ce0ea..00000000 --- a/extension/src/platform/SphereLib/spherepack.cpp +++ /dev/null @@ -1,138 +0,0 @@ -// Platform skeleton for SphereLib/spherepack.h (40250 SphereLib/spherepack.cpp), generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. -#include "SphereLib/StdAfx.h" -#include "SphereLib/spherepack.h" - -#include "../PlatformStub.h" - -auto SpherePack::LostChild(SpherePack *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePack::Render(unsigned int) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePack::Recompute(float) -> bool -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto SpherePack::VisibilityTest(const Frustum &, SpherePackCallback *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePack::RayTrace(const Vector3d &, const Vector3d &, float, SpherePackCallback *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePack::RangeTest(const Vector3d &, float, SpherePackCallback *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePack::PointTest2d(const Vector3d &, SpherePackCallback *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePack::Reset() -> void -{ - MT_PLATFORM_STUB(); -} - -SpherePackFactory::SpherePackFactory(int, float, float, float) -{ - MT_PLATFORM_STUB(); -} - -SpherePackFactory::~SpherePackFactory() -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::Process() -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::AddSphere_(const Vector3d &, float, void *, bool, int) -> SpherePack * -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} - -auto SpherePackFactory::AddIntegrate(SpherePack *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::AddRecompute(SpherePack *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::Integrate(SpherePack *, SpherePack *, float) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::Render() -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::Remove(SpherePack *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::FrustumTest(const Frustum &, SpherePackCallback *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::RayTrace(const Vector3d &, const Vector3d &, SpherePackCallback *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::RangeTest(const Vector3d &, float, SpherePackCallback *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::PointTest2d(const Vector3d &, SpherePackCallback *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::RayTraceCallback(const Vector3d &, const Vector3d &, float, const Vector3d &, SpherePack *) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::RangeTestCallback(const Vector3d &, float, SpherePack *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::PointTest2dCallback(const Vector3d &, SpherePack *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::VisibilityCallback(const Frustum &, SpherePack *, ViewState) -> void -{ - MT_PLATFORM_STUB(); -} - -auto SpherePackFactory::Reset() -> void -{ - MT_PLATFORM_STUB(); -} diff --git a/extension/src/platform/SphereLib/vector.cpp b/extension/src/platform/SphereLib/vector.cpp deleted file mode 100644 index d65f6c0e..00000000 --- a/extension/src/platform/SphereLib/vector.cpp +++ /dev/null @@ -1,12 +0,0 @@ -// Platform skeleton for SphereLib/vector.h, generated by platform_stub.py. -// Every MT_PLATFORM_STUB() body is unimplemented: replace it with the platform implementation. -#include "SphereLib/StdAfx.h" -#include "SphereLib/vector.h" - -#include "../PlatformStub.h" - -auto Vector3d::IsInStaticRange() const -> bool -{ - MT_PLATFORM_STUB(); - return mt_platform_stub_return(); -} diff --git a/extension/src/platform/UserInterface/PythonApplication.cpp b/extension/src/platform/UserInterface/PythonApplication.cpp index c8f8db20..176d8cea 100644 --- a/extension/src/platform/UserInterface/PythonApplication.cpp +++ b/extension/src/platform/UserInterface/PythonApplication.cpp @@ -25,8 +25,9 @@ // // PORT: there is no CPythonApplication object yet (its members are the whole game: background, // network stream, player, ...), so these functions only use the singletons PythonBoot owns -// (UI::CWindowManager, CTimer, CResourceManager) and never touch a data member. -CPythonApplication* CPythonApplication::ms_pInstance = nullptr; +alignas(alignof(CPythonApplication)) static char s_app_storage[sizeof(CPythonApplication)] = {}; +CPythonApplication* CPythonApplication::ms_pInstance = reinterpret_cast(s_app_storage); +extern double g_specularSpd; void CPythonApplication::ShowWebPage(const char*, const RECT&) { MT_PLATFORM_STUB(); } void CPythonApplication::MoveWebPage(const RECT&) { MT_PLATFORM_STUB(); } @@ -134,6 +135,16 @@ bool CPythonApplication::Create(PyObject*, const char*, int, int, int) // 40250 PythonApplication.cpp:1260. CPythonTextTail::Instance().Initialize(); + + if (IsNativeTerrainRenderEnabled()) + { + const bool use_hardware_cursor = PythonBoot::HostHardwareCursorEnabled(); + if (CPythonSystem::InstancePtr() && CPythonSystem::Instance().GetConfig()) + CPythonSystem::Instance().GetConfig()->is_software_cursor = !use_hardware_cursor; + SetCursorMode(use_hardware_cursor ? CURSOR_MODE_HARDWARE : CURSOR_MODE_SOFTWARE); + CGrannyMaterial::CreateSphereMap(0, "d:/ymir work/special/spheremap.jpg"); + CGrannyMaterial::CreateSphereMap(1, "d:/ymir work/special/spheremap01.jpg"); + } return true; } // 40250 PythonApplication.cpp:105 @@ -216,6 +227,7 @@ void CPythonApplication::UpdateGame() s.SetPerspective(30.0f,fAspect, 100.0f, fFarClip); s.BuildViewFrustum(); + s.SetCursorPosition(ptMouse.x, ptMouse.y, UI::CWindowManager::Instance().GetScreenWidth(), UI::CWindowManager::Instance().GetScreenHeight()); } TPixelPosition kPPosMainActor; @@ -262,8 +274,13 @@ bool CPythonApplication::Process() CResourceManager::Instance().Update(); OnCameraUpdate(); + if (IsNativeTerrainRenderEnabled()) + OnMouseUpdate(); OnUIUpdate(); + if (IsNativeTerrainRenderEnabled()) + CGrannyMaterial::TranslateSpecularMatrix(g_specularSpd, g_specularSpd, 0.0f); + // PORT: the render block without the lost-device restore, ClearDepthBuffer, Show and the render-time // statistics (Godot clears and presents the frame). The frame's UI and 3D command lists restart here. CCullingManager::Instance().Update(); @@ -275,6 +292,14 @@ bool CPythonApplication::Process() rkGraphic.SetInterfaceRenderState(); OnUIRender(); + if (IsNativeTerrainRenderEnabled()) + { + rkGraphic.SetInterfaceRenderState(); + unsigned cw = 0, ch = 0; + UIRenderGetSize(&cw, &ch); + UIRenderSetClip(0.0f, 0.0f, float(cw), float(ch)); + OnMouseRender(); + } rkGraphic.End(); } diff --git a/extension/src/platform/UserInterface/PythonApplicationEvent.cpp b/extension/src/platform/UserInterface/PythonApplicationEvent.cpp index 297acc00..f6a91c2b 100644 --- a/extension/src/platform/UserInterface/PythonApplicationEvent.cpp +++ b/extension/src/platform/UserInterface/PythonApplicationEvent.cpp @@ -11,7 +11,10 @@ void CPythonApplication::OnCameraUpdate() { CCamera* pkCameraMgr = CCameraManager::Instance().GetCurrentCamera(); if (pkCameraMgr) + { pkCameraMgr->Update(); + CPythonGraphic::Instance().UpdateViewMatrix(); + } } } diff --git a/extension/src/port/EterGrnLib/ModelInstanceCollisionDetection.cpp b/extension/src/port/EterGrnLib/ModelInstanceCollisionDetection.cpp index 5d96e293..06395bc8 100644 --- a/extension/src/port/EterGrnLib/ModelInstanceCollisionDetection.cpp +++ b/extension/src/port/EterGrnLib/ModelInstanceCollisionDetection.cpp @@ -9,31 +9,41 @@ void CGrannyModelInstance::MakeBoundBox(TBoundBox* pBoundBox, D3DXVECTOR3* vtMin, D3DXVECTOR3* vtMax) { - pBoundBox->sx = OBBMin[0] * mat[0] + OBBMin[1] * mat[4] + OBBMin[2] * mat[8] + mat[12]; - pBoundBox->sy = OBBMin[0] * mat[1] + OBBMin[1] * mat[5] + OBBMin[2] * mat[9] + mat[13]; - pBoundBox->sz = OBBMin[0] * mat[2] + OBBMin[1] * mat[6] + OBBMin[2] * mat[10] + mat[14]; + pBoundBox->sx = +10000000.0f; + pBoundBox->sy = +10000000.0f; + pBoundBox->sz = +10000000.0f; + pBoundBox->ex = -10000000.0f; + pBoundBox->ey = -10000000.0f; + pBoundBox->ez = -10000000.0f; - pBoundBox->ex = OBBMax[0] * mat[0] + OBBMax[1] * mat[4] + OBBMax[2] * mat[8] + mat[12]; - pBoundBox->ey = OBBMax[0] * mat[1] + OBBMax[1] * mat[5] + OBBMax[2] * mat[9] + mat[13]; - pBoundBox->ez = OBBMax[0] * mat[2] + OBBMax[1] * mat[6] + OBBMax[2] * mat[10] + mat[14]; + for (int c = 0; c < 8; ++c) + { + const float ox = (c & 1) ? OBBMax[0] : OBBMin[0]; + const float oy = (c & 2) ? OBBMax[1] : OBBMin[1]; + const float oz = (c & 4) ? OBBMax[2] : OBBMin[2]; + const float tx = ox * mat[0] + oy * mat[4] + oz * mat[8] + mat[12]; + const float ty = ox * mat[1] + oy * mat[5] + oz * mat[9] + mat[13]; + const float tz = ox * mat[2] + oy * mat[6] + oz * mat[10] + mat[14]; + + pBoundBox->sx = min(pBoundBox->sx, tx); + pBoundBox->sy = min(pBoundBox->sy, ty); + pBoundBox->sz = min(pBoundBox->sz, tz); + pBoundBox->ex = max(pBoundBox->ex, tx); + pBoundBox->ey = max(pBoundBox->ey, ty); + pBoundBox->ez = max(pBoundBox->ez, tz); + } vtMin->x = min(vtMin->x, pBoundBox->sx); - vtMin->x = min(vtMin->x, pBoundBox->ex); vtMin->y = min(vtMin->y, pBoundBox->sy); - vtMin->y = min(vtMin->y, pBoundBox->ey); vtMin->z = min(vtMin->z, pBoundBox->sz); - vtMin->z = min(vtMin->z, pBoundBox->ez); - vtMax->x = max(vtMax->x, pBoundBox->sx); vtMax->x = max(vtMax->x, pBoundBox->ex); - vtMax->y = max(vtMax->y, pBoundBox->sy); vtMax->y = max(vtMax->y, pBoundBox->ey); - vtMax->z = max(vtMax->z, pBoundBox->sz); vtMax->z = max(vtMax->z, pBoundBox->ez); } bool CGrannyModelInstance::Intersect(const D3DXMATRIX * c_pMatrix, - float * /*pu*/, float * /*pv*/, float * pt) + float * pu, float * pv, float * pt) { if (!m_pgrnModelInstance) return false; @@ -88,7 +98,51 @@ bool CGrannyModelInstance::Intersect(const D3DXMATRIX * c_pMatrix, return ret; } - return true; + bool hasVolumeBone = false; + for (int m = 0; m < meshCount; ++m) + { + const granny_mesh * pgrnMesh = m_pModel->GetGrannyModelPointer()->MeshBindings[m].Mesh; + int * boneIndices = __GetMeshBoneIndices(m); + + for (int b = 0; b < pgrnMesh->BoneBindingCount; ++b) + { + const granny_bone_binding& rgrnBoneBinding = pgrnMesh->BoneBindings[b]; + if (rgrnBoneBinding.OBBMin[0] >= rgrnBoneBinding.OBBMax[0] && + rgrnBoneBinding.OBBMin[1] >= rgrnBoneBinding.OBBMax[1] && + rgrnBoneBinding.OBBMin[2] >= rgrnBoneBinding.OBBMax[2]) + continue; + + hasVolumeBone = true; + const D3DXMATRIX * pBoneMat = (const D3DXMATRIX *)GrannyGetWorldPose4x4(__GetWorldPosePtr(), boneIndices[b]); + D3DXMATRIX matBoneWorld; + if (c_pMatrix) + D3DXMatrixMultiply(&matBoneWorld, pBoneMat, c_pMatrix); + else + matBoneWorld = *pBoneMat; + + if (IntersectCube(&matBoneWorld, + rgrnBoneBinding.OBBMin[0], rgrnBoneBinding.OBBMin[1], rgrnBoneBinding.OBBMin[2], + rgrnBoneBinding.OBBMax[0], rgrnBoneBinding.OBBMax[1], rgrnBoneBinding.OBBMax[2], + ms_vtPickRayOrig, ms_vtPickRayDir, + &u, &v, &t)) + { + if (pu) *pu = u; + if (pv) *pv = v; + if (pt) *pt = t; + return true; + } + } + } + + if (!hasVolumeBone) + { + if (pu) *pu = u; + if (pv) *pv = v; + if (pt) *pt = t; + return true; + } + + return false; /* TBoundBox* boundBoxs = s_boundBoxPool.base(); diff --git a/extension/src/port/EterLib/Camera.cpp b/extension/src/port/EterLib/Camera.cpp index cf929cfb..3b394e75 100644 --- a/extension/src/port/EterLib/Camera.cpp +++ b/extension/src/port/EterLib/Camera.cpp @@ -92,7 +92,16 @@ void CCamera::Wheel(int nLen) if (IsLock()) return; - m_v3AngularVelocity.y = (float)(nLen) * m_fResistance; + float fAdd = (float)(nLen) * m_fResistance; + if ((m_v3AngularVelocity.y > 0.0f && fAdd < 0.0f) || (m_v3AngularVelocity.y < 0.0f && fAdd > 0.0f)) + m_v3AngularVelocity.y = fAdd; + else + m_v3AngularVelocity.y += fAdd; + + if (m_v3AngularVelocity.y > 500.0f) + m_v3AngularVelocity.y = 500.0f; + else if (m_v3AngularVelocity.y < -500.0f) + m_v3AngularVelocity.y = -500.0f; } void CCamera::BeginDrag(int nMouseX, int nMouseY) @@ -167,6 +176,8 @@ bool CCamera::Drag(int nMouseX, int nMouseY, LPPOINT lpReturnPoint) m_v3AngularVelocity.x = fNewRotationVelocity; m_v3AngularVelocity.z = fNewPitchVelocity; + m_lMousePosX = lMouseX; + m_lMousePosY = lMouseY; lpReturnPoint->x = m_lMousePosX; lpReturnPoint->y = m_lMousePosY; return true; diff --git a/extension/src/port/EterLib/CullingManager.cpp b/extension/src/port/EterLib/CullingManager.cpp new file mode 100644 index 00000000..25ff1efe --- /dev/null +++ b/extension/src/port/EterLib/CullingManager.cpp @@ -0,0 +1,176 @@ +#include "StdAfx.h" +#include "CullingManager.h" +#include "GrpObjectInstance.h" + +//#define COUNT_SHOWING_SPHERE + +#ifdef COUNT_SHOWING_SPHERE +int showingcount = 0; +#endif + +void CCullingManager::RayTraceCallback(const Vector3d &/*p1*/, // source pos of ray + const Vector3d &/*dir*/, // dest pos of ray + float distance, + const Vector3d &/*sect*/, + SpherePack *sphere) +{ + //if (state!=VS_OUTSIDE) + //{ + if (m_RayFarDistance<=0.0f || m_RayFarDistance>=distance) + { +#ifdef SPHERELIB_STRICT + if (sphere->IS_SPHERE) + puts("CCullingManager::RayTraceCallback"); +#endif + m_list.push_back((CGraphicObjectInstance *)sphere->GetUserData()); + } + //f((CGraphicObjectInstance *)sphere->GetUserData()); + //} +} + + +void CCullingManager::VisibilityCallback(const Frustum &/*f*/,SpherePack *sphere,ViewState state) +{ +#ifdef SPHERELIB_STRICT + if (sphere->IS_SPHERE) + puts("CCullingManager::VisibilityCallback"); +#endif + + CGraphicObjectInstance * pInstance = (CGraphicObjectInstance*)sphere->GetUserData(); + /*if (state == VS_PARTIAL) + { + Vector3d v; + float r; + pInstance->GetBoundingSphere(v,r); + state = f.ViewVolumeTest(v,r); + }*/ + if (state == VS_OUTSIDE) + { +#ifdef COUNT_SHOWING_SPHERE + if (pInstance->isShow()) + { + Tracef("SH : %p ",sphere->GetUserData()); + showingcount--; + Tracef("show size : %5d\n",showingcount); + } + +#endif + pInstance->Hide(); + } + else + { +#ifdef COUNT_SHOWING_SPHERE + if (!pInstance->isShow()) + { + Tracef("HS : %p ",sphere->GetUserData()); + showingcount++; + Tracef("show size : %5d\n",showingcount); + } +#endif + pInstance->Show(); + } +} + +void CCullingManager::RangeTestCallback(const Vector3d &/*p*/,float /*distance*/,SpherePack *sphere,ViewState state) +{ +#ifdef SPHERELIB_STRICT + if (sphere->IS_SPHERE) + puts("CCullingManager::RangeTestCallback"); +#endif + if (state!=VS_OUTSIDE) + { + m_list.push_back((CGraphicObjectInstance *)sphere->GetUserData()); + //f((CGraphicObjectInstance *)sphere->GetUserData()); + } + //assert(false && "NOT REACHED"); +} + +void CCullingManager::Reset() +{ + m_Factory->Reset(); +} + +void CCullingManager::Update() +{ + // TODO : update each object + // ÇÏÁö¸»°í °¢ÀÚ ÇÏ°Ô ÇØº¸ÀÚ + + //DWORD time = ELTimer_GetMSec(); + //Reset(); + + m_Factory->Process(); + //Tracef("cull update : %3d ",ELTimer_GetMSec()-time); +} + +void CCullingManager::Process() +{ + //DWORD time = ELTimer_GetMSec(); + //Frustum f; + UpdateViewMatrix(); + UpdateProjMatrix(); + BuildViewFrustum(); + m_Factory->FrustumTest(GetFrustum(), this); + //Tracef("cull process : %3d ",ELTimer_GetMSec()-time); +} + +CCullingManager::CullingHandle CCullingManager::Register(CGraphicObjectInstance * obj) +{ + assert(obj); +#ifdef COUNT_SHOWING_SPHERE + Tracef("CR : %p ",obj); + showingcount++; + Tracef("show size : %5d\n",showingcount); +#endif + Vector3d center; + float radius; + obj->GetBoundingSphere(center,radius); + return m_Factory->AddSphere_(center,radius,obj, false); +} + +void CCullingManager::Unregister(CullingHandle h) +{ +#ifdef COUNT_SHOWING_SPHERE + if (((CGraphicObjectInstance*)h->GetUserData())->isShow()) + { + Tracef("DE : %p ",h->GetUserData()); + showingcount--; + Tracef("show size : %5d\n",showingcount); + } +#endif + m_Factory->Remove(h); +} + +CCullingManager::CCullingManager() +{ + m_Factory = new SpherePackFactory( + 10000, // maximum count + 6400, // root radius + 1600, // leaf radius + 400 // extra radius + ); +} + +CCullingManager::~CCullingManager() +{ + delete m_Factory; +} + +void CCullingManager::FindRange(const Vector3d &p, float radius) +{ + m_list.clear(); + m_Factory->RangeTest(p, radius, this); +} + +void CCullingManager::FindRay(const Vector3d &p1, const Vector3d &dir) +{ + m_RayFarDistance = -1; + m_list.clear(); + m_Factory->RayTrace(p1,dir,this); +} + +void CCullingManager::FindRayDistance(const Vector3d &p1, const Vector3d &dir, float distance) +{ + m_RayFarDistance = distance; + m_list.clear(); + m_Factory->RayTrace(p1,dir,this); +} \ No newline at end of file diff --git a/extension/src/port/EterLib/GrpBase.cpp b/extension/src/port/EterLib/GrpBase.cpp index 4a40c631..487a2119 100644 --- a/extension/src/port/EterLib/GrpBase.cpp +++ b/extension/src/port/EterLib/GrpBase.cpp @@ -139,6 +139,10 @@ void CGraphicBase::SetBackBufferSize(UINT uWidth, UINT uHeight) { ms_d3dPresentParameter.BackBufferWidth = uWidth; ms_d3dPresentParameter.BackBufferHeight = uHeight; + ms_iWidth = uWidth; + ms_iHeight = uHeight; + ms_Viewport.Width = uWidth; + ms_Viewport.Height = uHeight; } void CGraphicBase::SetDefaultIndexBuffer(UINT eDefIB) diff --git a/extension/src/port/EterPythonLib/PythonWindowManager.cpp b/extension/src/port/EterPythonLib/PythonWindowManager.cpp index 2e5c3ccf..2d76203f 100644 --- a/extension/src/port/EterPythonLib/PythonWindowManager.cpp +++ b/extension/src/port/EterPythonLib/PythonWindowManager.cpp @@ -723,13 +723,32 @@ namespace UI CWindow * pLayer = *ritor; CWindow * pPickedWindow = pLayer->PickWindow(x, y); - if (pPickedWindow != pLayer) + if (pPickedWindow != pLayer) { return pPickedWindow; + } } return NULL; } + bool CWindowManager::IsPointInsideActiveUI(long x, long y) + { + CWindow * pPicked = __PickWindow(x, y); + if (!pPicked || pPicked == m_pRootWindow) + return false; + + CWindow * pCur = pPicked; + while (pCur->GetParent() && pCur->GetParent() != m_pRootWindow) + { + pCur = pCur->GetParent(); + } + + if (pCur && (0 == strcmp(pCur->GetName(), "GAME") || pCur == m_pRootWindow)) + return false; + + return true; + } + void CWindowManager::SetMousePosition(long x, long y) { if (m_iHres==0) @@ -760,6 +779,13 @@ namespace UI SetMousePosition(x, y); CWindow * pPointWindow = __PickWindow(m_lMouseX, m_lMouseY); + if (x == 470 && y == 380) { + std::printf(">>> C++ RunMouseMove(470, 380): m_lMouse=(%ld, %ld), pPointWindow=%s (type=%s, is_show=%d)\n", + m_lMouseX, m_lMouseY, + pPointWindow ? pPointWindow->GetName() : "NULL", + pPointWindow ? typeid(*pPointWindow).name() : "none", + pPointWindow ? pPointWindow->IsShow() : 0); + } if (g_bShowOverInWindowName) { diff --git a/extension/src/port/EterPythonLib/PythonWindowManager.h b/extension/src/port/EterPythonLib/PythonWindowManager.h index d7681a21..7dad56a7 100644 --- a/extension/src/port/EterPythonLib/PythonWindowManager.h +++ b/extension/src/port/EterPythonLib/PythonWindowManager.h @@ -103,6 +103,9 @@ namespace UI void ActivateWindow(CWindow * pWin); void DeactivateWindow(); CWindow * GetActivateWindow(); + + bool IsPointInsideActiveUI(long x, long y); + CWindow * PickWindow(long x, long y) { return __PickWindow(x, y); } void SetTop(CWindow * pWin); void SetTopUIWindow(); void ResetCapture(); diff --git a/extension/src/port/GameLib/ActorInstance.cpp b/extension/src/port/GameLib/ActorInstance.cpp index bd66f0d9..c31f87d4 100644 --- a/extension/src/port/GameLib/ActorInstance.cpp +++ b/extension/src/port/GameLib/ActorInstance.cpp @@ -250,7 +250,7 @@ bool CActorInstance::IsHandMode() bool CActorInstance::IsTwoHandMode() { - if (CRaceMotionData::MODE_TWOHAND_SWORD==GetMotionMode()) + if (CRaceMotionData::MODE_TWOHAND_SWORD==GetMotionMode() || CRaceMotionData::MODE_HORSE_TWOHAND_SWORD==GetMotionMode()) return true; return false; diff --git a/extension/src/port/SphereLib/StdAfx.cpp b/extension/src/port/SphereLib/StdAfx.cpp new file mode 100644 index 00000000..20799a35 --- /dev/null +++ b/extension/src/port/SphereLib/StdAfx.cpp @@ -0,0 +1,6 @@ +// stdafx.cpp : source file that includes just the standard includes +// SphereLib.pch will be the pre-compiled header +// stdafx.obj will contain the pre-compiled type information + +#include "StdAfx.h" + diff --git a/extension/src/port/SphereLib/frustum.cpp b/extension/src/port/SphereLib/frustum.cpp new file mode 100644 index 00000000..3af65ee8 --- /dev/null +++ b/extension/src/port/SphereLib/frustum.cpp @@ -0,0 +1,93 @@ +/* Copyright (C) John W. Ratcliff, 2001. + * All rights reserved worldwide. + * + * This software is provided "as is" without express or implied + * warranties. You may freely copy and compile this source into + * applications you distribute provided that the copyright text + * below is included in the resulting source code, for example: + * "Portions Copyright (C) John W. Ratcliff, 2001" + */ + +#include "StdAfx.h" +#include "frustum.h" + +//#include "frustum.h" + +/*void Frustum::Set(int x1,int y1,int x2,int y2) +{ + mX1 = x1; + mY1 = y1; + mX2 = x2; + mY2 = y2; +} + +*/ +ViewState Frustum::ViewVolumeTest(const Vector3d &c_v3Center,const float c_fRadius) const +{ + if (m_bUsingSphere) + { + D3DXVECTOR3 v( + c_v3Center.x-m_v3Center.x, + c_v3Center.y-m_v3Center.y, + c_v3Center.z-m_v3Center.z); + + if ((c_fRadius + m_fRadius) * (c_fRadius + m_fRadius) < D3DXVec3LengthSq(&v)) + { + return VS_OUTSIDE; + } + } + + const int count=6; + + D3DXVECTOR3 center = c_v3Center; + //center.y *=-1; + + int i; + + float distance[count]; + for(i=0;i +#include +#include +#include + +bool Vector3d::IsInStaticRange() const +{ + const float LIMIT = 3276700.0f; + if (x-LIMIT) + if (y-LIMIT) + if (z-LIMIT) + return true; + + return false; +} + +void Sphere::Set(const Vector3d ¢er, float radius) +{ +#ifdef __STATIC_RANGE__ + assert(center.IsInStaticRange()); +#endif + mCenter = center; + mRadius = radius; + mRadius2 = radius*radius; +} + + +//ray-sphere intersection test from Graphics Gems p.388 +// **NOTE** There is a bug in this Graphics Gem. If the origin +// of the ray is *inside* the sphere being tested, it reports the +// wrong intersection location. This code has a fix for the bug. +bool Sphere::RayIntersection(const Vector3d &rayOrigin, + const Vector3d &dir, + Vector3d *intersect) +{ + //notation: + //point E = rayOrigin + //point O = sphere center + + Vector3d EO = mCenter - rayOrigin; + Vector3d V = dir; + float dist2 = EO.x*EO.x + EO.y*EO.y + EO.z * EO.z; + // Bug Fix For Gem, if origin is *inside* the sphere, invert the + // direction vector so that we get a valid intersection location. + if ( dist2 < mRadius2 ) V*=-1; + + float v = EO.Dot(V); + + float disc = mRadius2 - (EO.Length2() - v*v); + + if (disc > 0.0f) + { + + if ( intersect ) + { + + float d = sqrtf(disc); + + //float dist2 = rayOrigin.DistanceSq(mCenter); + + *intersect = rayOrigin + V*(v-d); + + } + + return true; + } + return false; +} + +// +bool Sphere::RayIntersection(const Vector3d &rayOrigin, + const Vector3d &V, + float distance, + Vector3d *intersect) +{ + Vector3d sect; + bool hit = RayIntersectionInFront(rayOrigin,V,§); + + if ( hit ) + { + float d = rayOrigin.DistanceSq(sect); + if ( d > (distance*distance) ) return false; + if ( intersect ) *intersect = sect; + return true; + } + return false; +} + + +bool Sphere::RayIntersectionInFront(const Vector3d &rayOrigin, + const Vector3d &V, + Vector3d *intersect) +{ + Vector3d sect; + bool hit = RayIntersection(rayOrigin,V,§); + + if ( hit ) + { + + Vector3d dir = sect - rayOrigin; + + float dot = dir.Dot(V); + + if ( dot >= 0 ) // then it's in front! + { + if ( intersect ) *intersect = sect; + return true; + } + } + return false; +} + +void Sphere::Report(void) +{ +} + +/* +An Efficient Bounding Sphere +by Jack Ritter +from "Graphics Gems", Academic Press, 1990 +*/ + +/* Routine to calculate tight bounding sphere over */ +/* a set of points in 3D */ +/* This contains the routine find_bounding_sphere(), */ +/* the struct definition, and the globals used for parameters. */ +/* The abs() of all coordinates must be < BIGNUMBER */ +/* Code written by Jack Ritter and Lyle Rains. */ + +#define BIGNUMBER 100000000.0 /* hundred million */ + +void Sphere::Compute(const SphereInterface &source) +{ + + Vector3d xmin,xmax,ymin,ymax,zmin,zmax,dia1,dia2; + + /* FIRST PASS: find 6 minima/maxima points */ + xmin.Set(BIGNUMBER,BIGNUMBER,BIGNUMBER); + xmax.Set(-BIGNUMBER,-BIGNUMBER,-BIGNUMBER); + ymin.Set(BIGNUMBER,BIGNUMBER,BIGNUMBER); + ymax.Set(-BIGNUMBER,-BIGNUMBER,-BIGNUMBER); + zmin.Set(BIGNUMBER,BIGNUMBER,BIGNUMBER); + zmax.Set(-BIGNUMBER,-BIGNUMBER,-BIGNUMBER); + + int count = source.GetVertexCount(); + + for (int i=0; ixmax.GetX()) xmax = caller_p; + if (caller_p.GetY()ymax.GetY()) ymax = caller_p; + if (caller_p.GetZ()zmax.GetZ()) zmax = caller_p; + } + + /* Set xspan = distance between the 2 points xmin & xmax (squared) */ + float dx = xmax.GetX() - xmin.GetX(); + float dy = xmax.GetY() - xmin.GetY(); + float dz = xmax.GetZ() - xmin.GetZ(); + float xspan = dx*dx + dy*dy + dz*dz; + + /* Same for y & z spans */ + dx = ymax.GetX() - ymin.GetX(); + dy = ymax.GetY() - ymin.GetY(); + dz = ymax.GetZ() - ymin.GetZ(); + float yspan = dx*dx + dy*dy + dz*dz; + + dx = zmax.GetX() - zmin.GetX(); + dy = zmax.GetY() - zmin.GetY(); + dz = zmax.GetZ() - zmin.GetZ(); + float zspan = dx*dx + dy*dy + dz*dz; + + /* Set points dia1 & dia2 to the maximally separated pair */ + dia1 = xmin; + dia2 = xmax; /* assume xspan biggest */ + float maxspan = xspan; + + if (yspan>maxspan) + { + maxspan = yspan; + dia1 = ymin; + dia2 = ymax; + } + + if (zspan>maxspan) + { + dia1 = zmin; + dia2 = zmax; + } + + + /* dia1,dia2 is a diameter of initial sphere */ + /* calc initial center */ + mCenter.SetX( (dia1.GetX()+dia2.GetX())*0.5f ); + mCenter.SetY( (dia1.GetY()+dia2.GetY())*0.5f ); + mCenter.SetZ( (dia1.GetZ()+dia2.GetZ())*0.5f ); + /* calculate initial radius**2 and radius */ + dx = dia2.GetX()-mCenter.GetX(); /* x component of radius vector */ + dy = dia2.GetY()-mCenter.GetY(); /* y component of radius vector */ + dz = dia2.GetZ()-mCenter.GetZ(); /* z component of radius vector */ + mRadius2 = dx*dx + dy*dy + dz*dz; + mRadius = float(sqrt(mRadius2)); + + /* SECOND PASS: increment current sphere */ + + for (int j=0; j mRadius2) /* do r**2 test first */ + { /* this point is outside of current sphere */ + float old_to_p = float(sqrt(old_to_p_sq)); + /* calc radius of new sphere */ + mRadius = (mRadius + old_to_p) * 0.5f; + mRadius2 = mRadius*mRadius; /* for next r**2 compare */ + float old_to_new = old_to_p - mRadius; + /* calc center of new sphere */ + float recip = 1.0f /old_to_p; + + float cx = (mRadius*mCenter.GetX() + old_to_new*caller_p.GetX()) * recip; + float cy = (mRadius*mCenter.GetY() + old_to_new*caller_p.GetY()) * recip; + float cz = (mRadius*mCenter.GetZ() + old_to_new*caller_p.GetZ()) * recip; + + mCenter.Set(cx,cy,cz); + } + } +} diff --git a/extension/src/port/SphereLib/spherepack.cpp b/extension/src/port/SphereLib/spherepack.cpp new file mode 100644 index 00000000..dab7759f --- /dev/null +++ b/extension/src/port/SphereLib/spherepack.cpp @@ -0,0 +1,878 @@ +/* Copyright (C) John W. Ratcliff, 2001. + * All rights reserved worldwide. + * + * This software is provided "as is" without express or implied + * warranties. You may freely copy and compile this source into + * applications you distribute provided that the copyright text + * below is included in the resulting source code, for example: + * "Portions Copyright (C) John W. Ratcliff, 2001" + */ + +#include "StdAfx.h" +#include "spherepack.h" + +#if DEMO +int PrintText(int x, int y, int color, char* output, ...); +int DrawLine(int x1, int y1, int x2, int y2, int color); +int DrawCircle(int locx, int locy, int radius, int color); +#endif + +SpherePackFactory::SpherePackFactory(int maxspheres, float rootsize, float leafsize, float gravy) +{ + NANOBEGIN + maxspheres *= 4; // include room for both trees, the root node and leaf node tree, and the superspheres + mMaxRootSize = rootsize; + mMaxLeafSize = leafsize; + mSuperSphereGravy = gravy; + mIntegrate = new SpherePackFifo(maxspheres); + mRecompute = new SpherePackFifo(maxspheres); + + mSpheres.Set(maxspheres); // init pool to hold all possible SpherePack instances. + + Vector3d p(0,0,0); + + mRoot = mSpheres.GetFreeLink(); // initially empty + mRoot->Init(this,p,6553600,0, false); + mRoot->SetSpherePackFlag(SpherePackFlag(SPF_SUPERSPHERE | SPF_ROOTNODE | SPF_ROOT_TREE)); + +#if DEMO + mRoot->SetColor(0x00FFFFFF); +#endif + + mLeaf = mSpheres.GetFreeLink();; // initially empty + mLeaf->Init(this,p,1638400,0,false); + mLeaf->SetSpherePackFlag(SpherePackFlag(SPF_SUPERSPHERE | SPF_ROOTNODE | SPF_LEAF_TREE)); + +#if DEMO + mLeaf->SetColor(0x00FFFFFF); + mColorCount = 0; + + mColors[0] = 0x00FF0000; + mColors[1] = 0x0000FF00; + mColors[2] = 0x000000FF; + mColors[3] = 0x00FFFF00; + mColors[4] = 0x00FF00FF; + mColors[5] = 0x0000FFFF; + mColors[6] = 0x00FF8080; + mColors[7] = 0x0000FF80; + mColors[8] = 0x000080FF; + mColors[9] = 0x00FFFF80; + mColors[10] = 0x00FF80FF; + mColors[11] = 0x0080FFFF; + +#endif + NANOEND +} + +SpherePackFactory::~SpherePackFactory(void) +{ + delete mIntegrate; // free up integration fifo + delete mRecompute; // free up recomputation fifo. +} + +void SpherePackFactory::Process(void) +{ + { + // First recompute anybody that needs to be recomputed!! + // When leaf node spheres exit their parent sphere, then the parent sphere needs to be rebalanced. In fact,it may now be empty and + // need to be removed. + // This is the location where (n) number of spheres in the recomputation FIFO are allowed to be rebalanced in the tree. + int maxrecompute = mRecompute->GetCount(); + for (int i = 0; i < maxrecompute; ++i) + { + SpherePack * pack = mRecompute->Pop(); + if (!pack) break; + pack->SetFifo1(0); // no longer on the fifo!! + bool kill = pack->Recompute(mSuperSphereGravy); + if (kill) Remove(pack); + } + } + + { + // Now, process the integration step. + int maxintegrate = mIntegrate->GetCount(); + + for (int i = 0; i < maxintegrate; ++i) + { + SpherePack * pack = mIntegrate->Pop(); + if (!pack) + break; + pack->SetFifo2(0); + + if (pack->HasSpherePackFlag(SPF_ROOT_TREE)) + Integrate(pack,mRoot,mMaxRootSize); // integrate this one single dude against the root node. + else + Integrate(pack,mLeaf,mMaxLeafSize); // integrate this one single dude against the root node. + } + } + +} + + +SpherePack * SpherePackFactory::AddSphere_(const Vector3d &pos, + float radius, + void *userdata, + bool isSphere, + int flags) +{ + + SpherePack *pack = mSpheres.GetFreeLink(); + + assert(pack); + + if (pack) + { + if (flags & SPF_ROOT_TREE) + { + pack->Init(this,pos,radius,userdata, isSphere); + pack->SetSpherePackFlag(SPF_ROOT_TREE); // member of the leaf node tree! + AddIntegrate(pack); // add to integration list. + } + else + { + pack->Init(this,pos,radius,userdata, isSphere); + pack->SetSpherePackFlag(SPF_LEAF_TREE); // member of the leaf node tree! + AddIntegrate(pack); // add to integration list. + } + } + + return pack; +} + +void SpherePackFactory::AddIntegrate(SpherePack *pack) +{ + if (pack->HasSpherePackFlag(SPF_ROOT_TREE)) + mRoot->AddChild(pack); + else + mLeaf->AddChild(pack); + + pack->SetSpherePackFlag(SPF_INTEGRATE); // still needs to be integrated! + SpherePack **fifo = mIntegrate->Push(pack); // add it to the integration stack. + pack->SetFifo2(fifo); +} + +void SpherePackFactory::AddRecompute(SpherePack *recompute) +{ + if (!recompute->HasSpherePackFlag(SPF_RECOMPUTE)) + { + if (recompute->GetChildCount()) + { + recompute->SetSpherePackFlag(SPF_RECOMPUTE); // needs to be recalculated! + SpherePack **fifo = mRecompute->Push(recompute); + recompute->SetFifo1(fifo); + } + else + { + Remove(recompute); + } + } +} + +void SpherePackFactory::Render(void) +{ +#if DEMO + mRoot->Render(mRoot->GetColor()); + mLeaf->Render(mLeaf->GetColor()); +#endif +} + + +void SpherePack::Render(unsigned int /*color*/) +{ +#if DEMO + if (!HasSpherePackFlag(SPF_ROOTNODE)) + { + + if (HasSpherePackFlag(SPF_SUPERSPHERE)) + { + color = mColor; + } + else + { + if (mParent->HasSpherePackFlag(SPF_ROOTNODE)) color = 0x00FFFFFF; + } +#if DEMO + DrawCircle(int(mCenter.x), int(mCenter.y),int(GetRadius()),color); +#endif + if (HasSpherePackFlag(SPF_SUPERSPHERE)) + { + if (HasSpherePackFlag(SPF_LEAF_TREE)) + { + +#if DEMO + DrawCircle(int(mCenter.x), int(mCenter.y),int(GetRadius()),color); +#endif +#ifdef SPHERELIB_STRICT + if (!sphere->IS_SPHERE) + puts("SpherePack::Render"); +#endif + SpherePack *link = (SpherePack *) GetUserData(); + + link = link->GetParent(); + + if (link && !link->HasSpherePackFlag(SPF_ROOTNODE)) + { + DrawLine(int(mCenter.x), int(mCenter.y), + int(link->mCenter.x), int(link->mCenter.y), + link->GetColor()); + } + } + else + { +#if DEMO + DrawCircle(int(mCenter.x), int(mCenter.y),int(GetRadius())+3,color); +#endif + } + + } + + } + + if (mChildren) + { + SpherePack *pack = mChildren; + + while (pack) + { + pack->Render(color); + pack = pack->_GetNextSibling(); + } + } +#endif +} + +bool SpherePack::Recompute(float gravy) +{ + if (!mChildren) return true; // kill it! + if (HasSpherePackFlag(SPF_ROOTNODE)) return false; // don't recompute root nodes! + +#if 1 + // recompute bounding sphere! + Vector3d total(0,0,0); + int count=0; + SpherePack *pack = mChildren; + while (pack) + { + total+=pack->mCenter; + count++; + pack = pack->_GetNextSibling(); + } + + if (count) + { + float recip = 1.0f / float(count); + total*=recip; + + Vector3d oldpos = mCenter; + +#ifdef __STATIC_RANGE__ + assert(total.IsInStaticRange()); +#endif + mCenter = total; // new origin! + float maxradius = 0; + + pack = mChildren; + + while (pack) + { + float dist = DistanceSquared(pack); + float radius = sqrtf(dist) + pack->GetRadius(); + if (radius > maxradius) + { + maxradius = radius; + if ((maxradius+gravy) >= GetRadius()) + { +#ifdef __STATIC_RANGE__ + assert(oldpos.IsInStaticRange()); +#endif + mCenter = oldpos; + ClearSpherePackFlag(SPF_RECOMPUTE); + return false; + } + } + pack = pack->_GetNextSibling(); + } + + maxradius+=gravy; + + SetRadius(maxradius); + + // now all children have to recompute binding distance!! + pack = mChildren; + + while (pack) + { + pack->ComputeBindingDistance(this); + pack = pack->_GetNextSibling(); + } + + } + +#endif + + ClearSpherePackFlag(SPF_RECOMPUTE); + + return false; +} + + +void SpherePack::LostChild(SpherePack *t) +{ + assert(mChildCount); + assert(mChildren); + +#ifdef _DEBUG // debug validation code. + + SpherePack *pack = mChildren; + bool found = false; + while (pack) + { + if (pack == t) + { + assert(!found); + found = true; + } + pack = pack->_GetNextSibling(); + } + assert(found); + +#endif + + // first patch old linked list.. his previous now points to his next + SpherePack *prev = t->_GetPrevSibling(); + + if (prev) + { + SpherePack *next = t->_GetNextSibling(); + prev->SetNextSibling(next); // my previous now points to my next + if (next) next->SetPrevSibling(prev); + // list is patched! + } + else + { + SpherePack *next = t->_GetNextSibling(); + mChildren = next; + if (mChildren) mChildren->SetPrevSibling(0); + } + + mChildCount--; + + if (!mChildCount && HasSpherePackFlag(SPF_SUPERSPHERE)) + { + mFactory->Remove(this); + } +} + +void SpherePackFactory::Remove(SpherePack*pack) +{ + + if (pack->HasSpherePackFlag(SPF_ROOTNODE)) return; // CAN NEVER REMOVE THE ROOT NODE EVER!!! + + if (pack->HasSpherePackFlag(SPF_SUPERSPHERE) && pack->HasSpherePackFlag(SPF_LEAF_TREE)) + { +#ifdef SPHERELIB_STRICT + if (!pack->IS_SPHERE) + puts("SpherePackFactory::Remove"); +#endif + SpherePack *link = (SpherePack *) pack->GetUserData(); + + Remove(link); + } + + pack->Unlink(); + + mSpheres.Release(pack); +} + +#if DEMO +unsigned int SpherePackFactory::GetColor(void) +{ + unsigned int ret = mColors[mColorCount]; + mColorCount++; + if (mColorCount == MAXCOLORS) mColorCount = 0; + return ret; +} +#endif + +void SpherePackFactory::Integrate(SpherePack *pack, + SpherePack *supersphere, + float node_size) +{ + // ok..time to integrate this sphere with the tree + // first find which supersphere we are closest to the center of + + SpherePack *search = supersphere->GetChildren(); + + SpherePack *nearest1 = 0; // nearest supersphere we are completely + float neardist1 = 1e38f; // enclosed within + + SpherePack *nearest2 = 0; // supersphere we must grow the least to + float neardist2 = 1e38f; // add ourselves to. + + //int scount = 1; + + while (search) + { + if (search->HasSpherePackFlag(SPF_SUPERSPHERE) && !search->HasSpherePackFlag(SPF_ROOTNODE) && search->GetChildCount()) + { + + float dist = pack->DistanceSquared(search); + + if (nearest1) + { + if (dist < neardist1) + { + + float d = sqrtf(dist)+pack->GetRadius(); + + if (d <= search->GetRadius()) + { + neardist1 = dist; + nearest1 = search; + } + } + } + else + { + + float d = (sqrtf(dist) + pack->GetRadius())-search->GetRadius(); + + if (d < neardist2) + { + if (d < 0) + { + neardist1 = dist; + nearest1 = search; + } + else + { + neardist2 = d; + nearest2 = search; + } + } + } + } + search = search->_GetNextSibling(); + } + + // ok...now..on exit let's see what we got. + if (nearest1) + { + // if we are inside an existing supersphere, we are all good! + // we need to detach item from wherever it is, and then add it to + // this supersphere as a child. + pack->Unlink(); + nearest1->AddChild(pack); + pack->ComputeBindingDistance(nearest1); + nearest1->Recompute(mSuperSphereGravy); + + if (nearest1->HasSpherePackFlag(SPF_LEAF_TREE)) + { +#ifdef SPHERELIB_STRICT + if (!nearest1->IS_SPHERE) + puts("SpherePackFactory::Integrate1"); +#endif + SpherePack *link = (SpherePack *) nearest1->GetUserData(); + link->NewPosRadius(nearest1->GetPos(), nearest1->GetRadius()); + } + + } + else + { + bool newsphere = true; + + if (nearest2) + { + float newsize = neardist2 + nearest2->GetRadius() + mSuperSphereGravy; + + if (newsize <= node_size) + { + pack->Unlink(); + + nearest2->SetRadius(newsize); + nearest2->AddChild(pack); + nearest2->Recompute(mSuperSphereGravy); + pack->ComputeBindingDistance(nearest2); + + if (nearest2->HasSpherePackFlag(SPF_LEAF_TREE)) + { +#ifdef SPHERELIB_STRICT + if (!nearest2->IS_SPHERE) + puts("SpherePackFactory::Integrate2"); +#endif + SpherePack *link = (SpherePack *) nearest2->GetUserData(); + link->NewPosRadius(nearest2->GetPos(), nearest2->GetRadius()); + } + + newsphere = false; + + } + + } + + if (newsphere) + { + assert(supersphere->HasSpherePackFlag(SPF_ROOTNODE)); + // we are going to create a new superesphere around this guy! + pack->Unlink(); + + SpherePack *parent = mSpheres.GetFreeLink(); + assert(parent); + parent->Init(this, pack->GetPos(), pack->GetRadius()+mSuperSphereGravy, 0, false); + + if (supersphere->HasSpherePackFlag(SPF_ROOT_TREE)) + parent->SetSpherePackFlag(SPF_ROOT_TREE); + else + parent->SetSpherePackFlag(SPF_LEAF_TREE); + + parent->SetSpherePackFlag(SPF_SUPERSPHERE); +#if DEMO + parent->SetColor(GetColor()); +#endif + parent->AddChild(pack); + + supersphere->AddChild(parent); + + parent->Recompute(mSuperSphereGravy); + pack->ComputeBindingDistance(parent); + + if (parent->HasSpherePackFlag(SPF_LEAF_TREE)) + { + // need to create parent association! + SpherePack *link = AddSphere_(parent->GetPos(), parent->GetRadius(), parent, true, SPF_ROOT_TREE); + parent->SetUserData(link, true); // hook him up!! + } + + } + } + + pack->ClearSpherePackFlag(SPF_INTEGRATE); // we've been integrated! +} + + +void SpherePackFactory::FrustumTest(const Frustum &f,SpherePackCallback *callback) +{ + // test case here, just traverse children. + mCallback = callback; + mRoot->VisibilityTest(f,this,VS_PARTIAL); +} + + +void SpherePack::VisibilityTest(const Frustum &f,SpherePackCallback *callback,ViewState state) +{ + + if (state == VS_PARTIAL) + { + state = f.ViewVolumeTest(mCenter, GetRadius()); +#if DEMO + if (state != VS_OUTSIDE) + { + DrawCircle(int(mCenter.x), int(mCenter.y), int(GetRadius()), 0x404040); + } +#endif + } + + if (HasSpherePackFlag(SPF_SUPERSPHERE)) + { + + + if (state == VS_OUTSIDE) + { + if (HasSpherePackFlag(SPF_HIDDEN)) return; // no state change + ClearSpherePackFlag(SpherePackFlag(SPF_INSIDE | SPF_PARTIAL)); + SetSpherePackFlag(SPF_HIDDEN); + } + else + { + if (state == VS_INSIDE) + { + if (HasSpherePackFlag(SPF_INSIDE)) return; // no state change + ClearSpherePackFlag(SpherePackFlag(SPF_PARTIAL | SPF_HIDDEN)); + SetSpherePackFlag(SPF_INSIDE); + } + else + { + ClearSpherePackFlag(SpherePackFlag(SPF_HIDDEN | SPF_INSIDE)); + SetSpherePackFlag(SPF_PARTIAL); + } + } + + SpherePack *pack = mChildren; + + while (pack) + { + pack->VisibilityTest(f,callback,state); + pack = pack->_GetNextSibling(); + } + + } + else + { + switch (state) + { + case VS_INSIDE: + if (!HasSpherePackFlag(SPF_INSIDE)) + { + ClearSpherePackFlag(SpherePackFlag(SPF_HIDDEN | SPF_PARTIAL)); + SetSpherePackFlag(SPF_INSIDE); + callback->VisibilityCallback(f,this,state); + } + break; + case VS_OUTSIDE: + if (!HasSpherePackFlag(SPF_HIDDEN)) + { + ClearSpherePackFlag(SpherePackFlag(SPF_INSIDE | SPF_PARTIAL)); + SetSpherePackFlag(SPF_HIDDEN); + callback->VisibilityCallback(f,this,state); + } + break; + case VS_PARTIAL: + //if (!HasSpherePackFlag(SPF_PARTIAL)) + { + ClearSpherePackFlag(SpherePackFlag(SPF_INSIDE | SPF_HIDDEN)); + SetSpherePackFlag(SPF_PARTIAL); + callback->VisibilityCallback(f,this,state); + } + break; + } + + } +} + +void SpherePackFactory::RayTrace(const Vector3d &p1, + const Vector3d &p2, + SpherePackCallback *callback) +{ + // test case here, just traverse children. + Vector3d dir = p2; + float dist = dir.Normalize(); + mCallback = callback; + mRoot->RayTrace(p1,dir,dist,this); +} + +#include "../EterBase/Debug.h" + +void SpherePackFactory::RangeTest(const Vector3d ¢er,float radius,SpherePackCallback *callback) +{ +#ifdef __STATIC_RANGE__ + if (!center.IsInStaticRange()) + { + TraceError("SpherePackFactory::RangeTest - RANGE ERROR %f, %f, %f", + center.x, center.y, center.z); + assert("SpherePackFactory::RangeTest - RANGE ERROR"); + return; + } +#endif + mCallback = callback; + mRoot->RangeTest(center,radius,this,VS_PARTIAL); +} + +void SpherePackFactory::PointTest2d(const Vector3d ¢er, SpherePackCallback *callback) +{ +#ifdef __STATIC_RANGE__ + if (!center.IsInStaticRange()) + { + TraceError("SpherePackFactory::RangeTest2d - RANGE ERROR %f, %f, %f", + center.x, center.y, center.z); + assert("SpherePackFactory::RangeTest2d - RANGE ERROR"); + return; + } +#endif + mCallback = callback; + +#ifdef SPHERELIB_STRICT + mRoot->PointTest2d(center, this,VS_PARTIAL); + extern bool MAPOUTDOOR_GET_HEIGHT_TRACE; + if (MAPOUTDOOR_GET_HEIGHT_TRACE) + puts("================================================"); +#else + mRoot->PointTest2d(center, this,VS_PARTIAL); + +#endif + +} + +void SpherePack::RangeTest(const Vector3d &p, + float distance, + SpherePackCallback *callback, + ViewState state) +{ + + if (state == VS_PARTIAL) + { + float d = p.Distance(mCenter); + if ((d-distance) > GetRadius()) return;; + if ((GetRadius()+d) < distance) state = VS_INSIDE; + } + + if (HasSpherePackFlag(SPF_SUPERSPHERE)) + { +#if DEMO + if (state == VS_PARTIAL) + { + DrawCircle(int(mCenter.x), int(mCenter.y), int(GetRadius()), 0x404040); + } +#endif + SpherePack *pack = mChildren; + while (pack) + { + pack->RangeTest(p,distance,callback,state); + pack = pack->_GetNextSibling(); + } + + } + else + { + callback->RangeTestCallback(p,distance,this,state); + } +} + +void SpherePack::PointTest2d(const Vector3d &p, + SpherePackCallback *callback, + ViewState state) +{ + if (state == VS_PARTIAL) + { + float dx=p.x-mCenter.x; + float dy=p.y-mCenter.y; + float distSquare = (dx*dx)+(dy*dy); + + if (distSquare > GetRadius2()) return;; + if (GetRadius2() < -distSquare) state = VS_INSIDE; + } + + if (HasSpherePackFlag(SPF_SUPERSPHERE)) + { +#if DEMO + if (state == VS_PARTIAL) + { + DrawCircle(int(mCenter.x), int(mCenter.y), int(GetRadius()), 0x404040); + } +#endif + SpherePack *pack = mChildren; + while (pack) + { + pack->PointTest2d(p, callback, state); + pack = pack->_GetNextSibling(); + } + + } + else + { +#ifdef SPHERELIB_STRICT + extern bool MAPOUTDOOR_GET_HEIGHT_TRACE; + if (MAPOUTDOOR_GET_HEIGHT_TRACE) + { + float dx=p.x-mCenter.x; + float dy=p.y-mCenter.y; + float distSquare = (dx*dx)+(dy*dy); + printf("--- (%f, %f) dist %f radius %f isSphere %d\n", mCenter.x, mCenter.y, distSquare, GetRadius(), IS_SPHERE); + } +#endif + callback->PointTest2dCallback(p, this, state); + } +} + +void SpherePackFactory::RangeTestCallback(const Vector3d &p,float distance,SpherePack *sphere,ViewState state) +{ +#ifdef SPHERELIB_STRICT + if (!sphere->IS_SPHERE) + puts("SpherePackFactory::RangeTestCallback"); +#endif + SpherePack *link = (SpherePack *) sphere->GetUserData(); + if (link) link->RangeTest(p,distance,mCallback,state); +}; + +void SpherePackFactory::PointTest2dCallback(const Vector3d &p, SpherePack *sphere,ViewState state) +{ +#ifdef SPHERELIB_STRICT + if (!sphere->IS_SPHERE) + puts("SpherePackFactory::PointTest2dCallback"); +#endif + SpherePack *link = (SpherePack *) sphere->GetUserData(); + if (link) link->PointTest2d(p, mCallback,state); +}; + +void SpherePack::RayTrace(const Vector3d &p1, + const Vector3d &dir, + float distance, + SpherePackCallback *callback) +{ + bool hit = false; + + if (HasSpherePackFlag(SPF_SUPERSPHERE)) + { + + hit = RayIntersectionInFront(p1,dir,0); + + if (hit) + { +#if DEMO + DrawCircle(int(mCenter.x), int(mCenter.y), int(GetRadius()), 0x404040); +#endif + SpherePack *pack = mChildren; + + while (pack) + { + pack->RayTrace(p1,dir,distance,callback); + pack = pack->_GetNextSibling(); + } + } + + } + else + { + Vector3d sect; + hit = RayIntersection(p1,dir,distance,§); + if (hit) + { + callback->RayTraceCallback(p1,dir,distance,sect,this); + } + } +} + +void SpherePackFactory::RayTraceCallback(const Vector3d &p1, // source pos of ray + const Vector3d &dir, // direction of ray + float distance, // distance of ray + const Vector3d &/*sect*/, // intersection location + SpherePack *sphere) +{ +#ifdef SPHERELIB_STRICT + if (!sphere->IS_SPHERE) + puts("SpherePackFactory::RayTraceCallback"); +#endif + SpherePack *link = (SpherePack *) sphere->GetUserData(); + if (link) link->RayTrace(p1,dir,distance,mCallback); +}; + + + + +void SpherePackFactory::Reset(void) +{ + mRoot->Reset(); + mLeaf->Reset(); +} + + +void SpherePack::Reset(void) +{ + ClearSpherePackFlag(SpherePackFlag(SPF_HIDDEN | SPF_PARTIAL | SPF_INSIDE)); + + SpherePack *pack = mChildren; + while (pack) + { + pack->Reset(); + pack = pack->_GetNextSibling(); + } +} + +void SpherePackFactory::VisibilityCallback(const Frustum &f,SpherePack *sphere,ViewState state) +{ +#ifdef SPHERELIB_STRICT + if (!sphere->IS_SPHERE) + puts("SpherePackFactory::VisibilityCallback"); +#endif + SpherePack *link = (SpherePack *) sphere->GetUserData(); + if (link) link->VisibilityTest(f,mCallback,state); +} diff --git a/extension/src/port/UserInterface/InstanceBase.cpp b/extension/src/port/UserInterface/InstanceBase.cpp index 2f126dd3..5eb75c42 100644 --- a/extension/src/port/UserInterface/InstanceBase.cpp +++ b/extension/src/port/UserInterface/InstanceBase.cpp @@ -2775,6 +2775,9 @@ void CInstanceBase::ChangeWeapon(DWORD eWeapon) if (SetWeapon(eWeapon)) RefreshState(CRaceMotionData::NAME_WAIT, true); + + if (IsAffect(AFFECT_GEOMGYEONG)) + __Warrior_SetGeomgyeongAffect(true); } bool CInstanceBase::ChangeArmor(DWORD dwArmor) @@ -3028,6 +3031,7 @@ void CInstanceBase::__Warrior_Initialize() void CInstanceBase::__Initialize() { __Warrior_Initialize(); + memset(m_byAffectGrade, 0, sizeof(m_byAffectGrade)); __StoneSmoke_Inialize(); __EffectContainer_Initialize(); __InitializeRotationSpeed(); diff --git a/extension/src/port/UserInterface/InstanceBase.h b/extension/src/port/UserInterface/InstanceBase.h index 7c3ea030..cc08a914 100644 --- a/extension/src/port/UserInterface/InstanceBase.h +++ b/extension/src/port/UserInterface/InstanceBase.h @@ -436,6 +436,7 @@ class CInstanceBase // 스크립트용 테스트 함수. 나중에 없에자 void SCRIPT_SetAffect(UINT eAffect, bool isVisible); + void __Warrior_SetGeomgyeongAffect(bool isVisible); float CalculateDistanceSq3d(const TPixelPosition& c_rkPPosDst); @@ -803,7 +804,7 @@ class CInstanceBase void __ClearMainInstance(); void __Shaman_SetParalysis(bool isParalysis); - void __Warrior_SetGeomgyeongAffect(bool isVisible); + BYTE __GetAffectGrade(UINT eAffect); void __Assassin_SetEunhyeongAffect(bool isVisible); void __SetReviveInvisibilityAffect(bool isVisible); @@ -1045,6 +1046,7 @@ class CInstanceBase }; SWarrior m_kWarrior; + BYTE m_byAffectGrade[AFFECT_NUM]; void __Warrior_Initialize(); diff --git a/extension/src/port/UserInterface/InstanceBaseBattle.cpp b/extension/src/port/UserInterface/InstanceBaseBattle.cpp index b551403b..3ac2e071 100644 --- a/extension/src/port/UserInterface/InstanceBaseBattle.cpp +++ b/extension/src/port/UserInterface/InstanceBaseBattle.cpp @@ -340,6 +340,66 @@ bool CInstanceBase::NEW_UseSkill(UINT uSkill, UINT uMot, UINT uMotLoopCount, boo m_GraphicThingInstance.__OnUseSkill(uMot, uMotLoopCount, isMovingSkill); + BYTE bGrade = (uMot >= 25) ? (uMot / 25) : 0; + if (bGrade > 3) + bGrade = 3; + + if (uSkill != 0) + { + switch (uSkill) + { + case 3: m_byAffectGrade[AFFECT_JEONGWI] = bGrade; break; + case 4: m_byAffectGrade[AFFECT_GEOMGYEONG] = bGrade; break; + case 19: m_byAffectGrade[AFFECT_CHEONGEUN] = bGrade; break; + case 34: m_byAffectGrade[AFFECT_EUNHYEONG] = bGrade; break; + case 49: m_byAffectGrade[AFFECT_GYEONGGONG] = bGrade; break; + case 63: m_byAffectGrade[AFFECT_GWIGEOM] = bGrade; break; + case 64: m_byAffectGrade[AFFECT_GONGPO] = bGrade; break; + case 65: m_byAffectGrade[AFFECT_JUMAGAP] = bGrade; break; + case 78: m_byAffectGrade[AFFECT_MUYEONG] = bGrade; break; + case 79: m_byAffectGrade[AFFECT_HEUKSIN] = bGrade; break; + case 94: m_byAffectGrade[AFFECT_HOSIN] = bGrade; break; + case 95: m_byAffectGrade[AFFECT_BOHO] = bGrade; break; + case 96: m_byAffectGrade[AFFECT_GICHEON] = bGrade; break; + case 110: m_byAffectGrade[AFFECT_KWAESOK] = bGrade; break; + case 111: m_byAffectGrade[AFFECT_JEUNGRYEOK] = bGrade; break; + } + } + else + { + UINT uMotSub = uMot % 25; + int iJob = RaceToJob(GetRace()); + switch (iJob) + { + case NRaceData::JOB_WARRIOR: + if (uMotSub == 3) m_byAffectGrade[AFFECT_JEONGWI] = bGrade; + else if (uMotSub == 4) m_byAffectGrade[AFFECT_GEOMGYEONG] = bGrade; + else if (uMotSub == 19) m_byAffectGrade[AFFECT_CHEONGEUN] = bGrade; + break; + case NRaceData::JOB_ASSASSIN: + if (uMotSub == 4) m_byAffectGrade[AFFECT_EUNHYEONG] = bGrade; + else if (uMotSub == 19) m_byAffectGrade[AFFECT_GYEONGGONG] = bGrade; + break; + case NRaceData::JOB_SURA: + if (uMotSub == 3) m_byAffectGrade[AFFECT_GWIGEOM] = bGrade; + else if (uMotSub == 4) m_byAffectGrade[AFFECT_GONGPO] = bGrade; + else if (uMotSub == 5) m_byAffectGrade[AFFECT_JUMAGAP] = bGrade; + else if (uMotSub == 18) m_byAffectGrade[AFFECT_MUYEONG] = bGrade; + else if (uMotSub == 19) m_byAffectGrade[AFFECT_HEUKSIN] = bGrade; + break; + case NRaceData::JOB_SHAMAN: + if (uMotSub == 4) m_byAffectGrade[AFFECT_HOSIN] = bGrade; + else if (uMotSub == 5) m_byAffectGrade[AFFECT_BOHO] = bGrade; + else if (uMotSub == 6) m_byAffectGrade[AFFECT_GICHEON] = bGrade; + else if (uMotSub == 20) m_byAffectGrade[AFFECT_KWAESOK] = bGrade; + else if (uMotSub == 21) m_byAffectGrade[AFFECT_JEUNGRYEOK] = bGrade; + break; + } + } + + if (IsAffect(AFFECT_GEOMGYEONG)) + __Warrior_SetGeomgyeongAffect(true); + if (uMotLoopCount > 0) m_GraphicThingInstance.SetMotionLoopCount(uMotLoopCount); diff --git a/extension/src/port/UserInterface/InstanceBaseEffect.cpp b/extension/src/port/UserInterface/InstanceBaseEffect.cpp index 08d9cc05..7b477a13 100644 --- a/extension/src/port/UserInterface/InstanceBaseEffect.cpp +++ b/extension/src/port/UserInterface/InstanceBaseEffect.cpp @@ -825,6 +825,29 @@ void CInstanceBase::__Shaman_SetParalysis(bool isParalysis) +BYTE CInstanceBase::__GetAffectGrade(UINT eAffect) +{ + if (__IsMainInstance()) + { + DWORD dwSkillIndex = 0; + if (CPythonPlayer::Instance().AffectIndexToSkillIndex(eAffect, &dwSkillIndex)) + { + DWORD dwSkillSlotIndex = 0; + if (CPythonPlayer::Instance().GetSkillSlotIndex(dwSkillIndex, &dwSkillSlotIndex)) + { + int iGrade = CPythonPlayer::Instance().GetSkillGrade(dwSkillSlotIndex); + if (iGrade >= 0 && iGrade < 4) + return (BYTE)iGrade; + } + } + } + + if (eAffect < AFFECT_NUM) + return m_byAffectGrade[eAffect]; + + return 0; +} + void CInstanceBase::__Warrior_SetGeomgyeongAffect(bool isVisible) { if (isVisible) @@ -836,10 +859,38 @@ void CInstanceBase::__Warrior_SetGeomgyeongAffect(bool isVisible) __DetachEffect(m_kWarrior.m_dwGeomgyeongEffect); m_GraphicThingInstance.SetReachScale(1.5f); - if (m_GraphicThingInstance.IsTwoHandMode()) - m_kWarrior.m_dwGeomgyeongEffect=__AttachEffect(EFFECT_WEAPON+WEAPON_TWOHAND); + + BYTE bGrade = __GetAffectGrade(AFFECT_GEOMGYEONG); + if (bGrade >= 4) + bGrade = 3; + + static const char * c_szGeomSpearFiles[4] = { + "d:/ymir work/pc/warrior/effect/geom_spear_loop.mse", + "d:/ymir work/pc/warrior/effect/geom_2_spear_loop.mse", + "d:/ymir work/pc/warrior/effect/geom_3_spear_loop.mse", + "d:/ymir work/pc/warrior/effect/geom_4_spear_loop.mse" + }; + static const char * c_szGeomSwordFiles[4] = { + "d:/ymir work/pc/warrior/effect/geom_sword_loop.mse", + "d:/ymir work/pc/warrior/effect/geom_2_sword_loop.mse", + "d:/ymir work/pc/warrior/effect/geom_3_sword_loop.mse", + "d:/ymir work/pc/warrior/effect/geom_4_sword_loop.mse" + }; + + const char * c_szFileName = m_GraphicThingInstance.IsTwoHandMode() ? c_szGeomSpearFiles[bGrade] : c_szGeomSwordFiles[bGrade]; + + DWORD dwEffectCRC = 0; + if (CEffectManager::Instance().RegisterEffect2(c_szFileName, &dwEffectCRC, true)) + { + m_kWarrior.m_dwGeomgyeongEffect = m_GraphicThingInstance.AttachEffectByID(0, "equip_right_hand", dwEffectCRC); + } else - m_kWarrior.m_dwGeomgyeongEffect=__AttachEffect(EFFECT_WEAPON+WEAPON_ONEHAND); + { + if (m_GraphicThingInstance.IsTwoHandMode()) + m_kWarrior.m_dwGeomgyeongEffect = __AttachEffect(EFFECT_WEAPON + WEAPON_TWOHAND); + else + m_kWarrior.m_dwGeomgyeongEffect = __AttachEffect(EFFECT_WEAPON + WEAPON_ONEHAND); + } } else { @@ -1033,6 +1084,129 @@ DWORD CInstanceBase::__AttachEffect(UINT eEftType) if (eEftType>=EFFECT_NUM) return 0; + if (eEftType >= EFFECT_AFFECT && eEftType < EFFECT_AFFECT_END) + { + UINT eAffect = eEftType - EFFECT_AFFECT; + BYTE bGrade = __GetAffectGrade(eAffect); + if (bGrade >= 4) + bGrade = 3; + + const char * c_szFileName = NULL; + const char * c_szBoneName = NULL; + + switch (eAffect) + { + case AFFECT_GWIGEOM: + { + static const char * c_szGwigeomFiles[4] = { + "d:/ymir work/pc/sura/effect/gwigeom_loop.mse", + "d:/ymir work/pc/sura/effect/gwigeom_2_loop.mse", + "d:/ymir work/pc/sura/effect/gwigeom_3_loop.mse", + "d:/ymir work/pc/sura/effect/gwigeom_4_loop.mse" + }; + c_szFileName = c_szGwigeomFiles[bGrade]; + c_szBoneName = "Bip01 R Finger2"; + break; + } + case AFFECT_GONGPO: + { + static const char * c_szFearFiles[4] = { + "d:/ymir work/pc/sura/effect/fear_loop.mse", + "d:/ymir work/pc/sura/effect/fear_2_loop.mse", + "d:/ymir work/pc/sura/effect/fear_3_loop.mse", + "d:/ymir work/pc/sura/effect/fear_3_loop.mse" + }; + c_szFileName = c_szFearFiles[bGrade]; + c_szBoneName = ""; + break; + } + case AFFECT_JUMAGAP: + { + static const char * c_szJumagapFiles[4] = { + "d:/ymir work/pc/sura/effect/jumagap_loop.mse", + "d:/ymir work/pc/sura/effect/jumagap_2_loop.mse", + "d:/ymir work/pc/sura/effect/jumagap_3_loop.mse", + "d:/ymir work/pc/sura/effect/jumagap_4_loop.mse" + }; + c_szFileName = c_szJumagapFiles[bGrade]; + c_szBoneName = ""; + break; + } + case AFFECT_HEUKSIN: + { + static const char * c_szHeuksinFiles[4] = { + "d:/ymir work/pc/sura/effect/heuksin_loop.mse", + "d:/ymir work/pc/sura/effect/heuksin_2_loop.mse", + "d:/ymir work/pc/sura/effect/heuksin_3_loop.mse", + "d:/ymir work/pc/sura/effect/heuksin_4_loop.mse" + }; + c_szFileName = c_szHeuksinFiles[bGrade]; + c_szBoneName = ""; + break; + } + case AFFECT_HOSIN: + { + c_szFileName = (bGrade >= 3) ? "d:/ymir work/pc/shaman/effect/3hosin_loop_4.mse" : "d:/ymir work/pc/shaman/effect/3hosin_loop.mse"; + c_szBoneName = ""; + break; + } + case AFFECT_BOHO: + { + c_szFileName = (bGrade >= 3) ? "d:/ymir work/pc/shaman/effect/boho_loop_4.mse" : "d:/ymir work/pc/shaman/effect/boho_loop.mse"; + c_szBoneName = ""; + break; + } + case AFFECT_GYEONGGONG: + { + static const char * c_szGyeonggongFiles[4] = { + "d:/ymir work/pc/assassin/effect/gyeonggong_loop.mse", + "d:/ymir work/pc/assassin/effect/gyeonggong_2_loop.mse", + "d:/ymir work/pc/assassin/effect/gyeonggong_3_loop.mse", + "d:/ymir work/pc/assassin/effect/gyeonggong_4_loop.mse" + }; + c_szFileName = c_szGyeonggongFiles[bGrade]; + c_szBoneName = ""; + break; + } + case AFFECT_CHEONGEUN: + { + static const char * c_szCheongeunFiles[4] = { + "d:/ymir work/pc/warrior/effect/gyeokgongjang_loop.mse", + "d:/ymir work/pc/warrior/effect/gyeokgongjang_2_loop.mse", + "d:/ymir work/pc/warrior/effect/gyeokgongjang_3_loop.mse", + "d:/ymir work/pc/warrior/effect/gyeokgongjang_3_loop.mse" + }; + c_szFileName = c_szCheongeunFiles[bGrade]; + c_szBoneName = ""; + break; + } + case AFFECT_GICHEON: + { + c_szFileName = (bGrade >= 3) ? "d:/ymir work/pc/shaman/effect/6gicheon_hand_4.mse" : "d:/ymir work/pc/shaman/effect/6gicheon_hand.mse"; + c_szBoneName = "Bip01 R Hand"; + break; + } + case AFFECT_JEUNGRYEOK: + { + c_szFileName = (bGrade >= 3) ? "d:/ymir work/pc/shaman/effect/jeungryeok_hand_4.mse" : "d:/ymir work/pc/shaman/effect/jeungryeok_hand.mse"; + c_szBoneName = "Bip01 L Hand"; + break; + } + default: + break; + } + + if (c_szFileName) + { + DWORD dwEffectCRC = 0; + if (CEffectManager::Instance().RegisterEffect2(c_szFileName, &dwEffectCRC, true)) + { + const char * bone = (c_szBoneName && c_szBoneName[0]) ? c_szBoneName : NULL; + return m_GraphicThingInstance.AttachEffectByID(0, bone, dwEffectCRC); + } + } + } + if (ms_astAffectEffectAttachBone[eEftType].empty()) { return m_GraphicThingInstance.AttachEffectByID(0, NULL, ms_adwCRCAffectEffect[eEftType]); diff --git a/extension/src/port/UserInterface/PythonApplication.h b/extension/src/port/UserInterface/PythonApplication.h index fadda877..841f3d88 100644 --- a/extension/src/port/UserInterface/PythonApplication.h +++ b/extension/src/port/UserInterface/PythonApplication.h @@ -274,17 +274,7 @@ class CPythonApplication final : public CMSApplication, public CInputKeyboard, p int m_nLeft, m_nRight, m_nTop, m_nBottom; - protected: - LRESULT WindowProcedure(HWND hWnd, UINT uiMsg, WPARAM wParam, LPARAM lParam); - - void OnCameraUpdate(); - - void OnUIUpdate(); - void OnUIRender(); - - void OnMouseUpdate(); - void OnMouseRender(); - + public: void OnMouseWheel(int nLen); void OnMouseMove(int x, int y); void OnMouseMiddleButtonDown(int x, int y); @@ -299,6 +289,17 @@ class CPythonApplication final : public CMSApplication, public CInputKeyboard, p void OnKeyUp(int iIndex); void OnIMEKeyDown(int iIndex); + protected: + LRESULT WindowProcedure(HWND hWnd, UINT uiMsg, WPARAM wParam, LPARAM lParam); + + void OnCameraUpdate(); + + void OnUIUpdate(); + void OnUIRender(); + + void OnMouseUpdate(); + void OnMouseRender(); + int CheckDeviceState(); BOOL __IsContinuousChangeTypeCursor(int iCursorNum); diff --git a/extension/src/port/UserInterface/PythonBackground.cpp b/extension/src/port/UserInterface/PythonBackground.cpp index bcf95dd4..87d7d5ee 100644 --- a/extension/src/port/UserInterface/PythonBackground.cpp +++ b/extension/src/port/UserInterface/PythonBackground.cpp @@ -254,10 +254,34 @@ void CPythonBackground::__CreateProperty() { // The exported client runs with player settings as CWD; its read-only property pack // lives beside the other bundled 40250 packs. Use the same root registered by PackBackend. - std::string property_pack = "pack/property"; + std::vector candidates; if (const char* client = getenv("MT_40250_CLIENT")) - property_pack = std::string(client) + "/pack/Property"; - m_PropertyManager.Initialize(property_pack.c_str()); + { + candidates.push_back(std::string(client) + "/pack/Property"); + candidates.push_back(std::string(client) + "/pack/property"); + } + candidates.push_back("pack/Property"); + candidates.push_back("pack/property"); + candidates.push_back("Client/pack/Property"); + candidates.push_back("Client/pack/property"); + + bool initialized = false; + for (const auto& path : candidates) + { + std::string eix = path + ".eix"; + if (_access(eix.c_str(), 0) == 0) + { + if (m_PropertyManager.Initialize(path.c_str())) + { + initialized = true; + break; + } + } + } + if (!initialized && !candidates.empty()) + { + m_PropertyManager.Initialize(candidates.front().c_str()); + } } } diff --git a/extension/src/port/UserInterface/PythonCharacterManager.cpp b/extension/src/port/UserInterface/PythonCharacterManager.cpp index 60d6b799..cd205183 100644 --- a/extension/src/port/UserInterface/PythonCharacterManager.cpp +++ b/extension/src/port/UserInterface/PythonCharacterManager.cpp @@ -735,6 +735,11 @@ void CPythonCharacterManager::__SortPickedActorList() std::sort(m_kVct_pkInstPicked.begin(), m_kVct_pkInstPicked.end(), kLess); } +void CPythonCharacterManager::Pick() +{ + __NEW_Pick(); +} + void CPythonCharacterManager::__NEW_Pick() { __UpdateSortPickedActorList(); @@ -773,31 +778,10 @@ void CPythonCharacterManager::__NEW_Pick() } } - // 못찾겠으면 걍 순서대로 - { - std::vector::iterator f; - for (f=m_kVct_pkInstPicked.begin(); f!=m_kVct_pkInstPicked.end(); ++f) - { - CInstanceBase* pkInstEach=*f; - if (pkInstEach!=pkInstMain) - { - if (m_pkInstPick) - if (m_pkInstPick!=pkInstEach) - m_pkInstPick->OnUnselected(); - - if (pkInstEach->CanPickInstance()) - { - m_pkInstPick = pkInstEach; - m_pkInstPick->OnSelected(); - return; - } - } - } - } - if (pkInstMain) if (pkInstMain->CanPickInstance()) if (m_kVct_pkInstPicked.end() != std::find(m_kVct_pkInstPicked.begin(), m_kVct_pkInstPicked.end(), pkInstMain)) + if (pkInstMain->IntersectBoundingBox()) { if (m_pkInstPick) if (m_pkInstPick!=pkInstMain) diff --git a/extension/src/port/UserInterface/PythonCharacterManager.h b/extension/src/port/UserInterface/PythonCharacterManager.h index ad41f4b0..31c3a02f 100644 --- a/extension/src/port/UserInterface/PythonCharacterManager.h +++ b/extension/src/port/UserInterface/PythonCharacterManager.h @@ -85,6 +85,7 @@ class CPythonCharacterManager : public CSingleton, publ CInstanceBase * GetInstancePtrByName(const char *name); // Pick + void Pick(); int PickAll(); CInstanceBase * GetCloseInstance(CInstanceBase * pInstance); diff --git a/extension/src/port/UserInterface/PythonPlayer.cpp b/extension/src/port/UserInterface/PythonPlayer.cpp index ec2f64ab..6f900171 100644 --- a/extension/src/port/UserInterface/PythonPlayer.cpp +++ b/extension/src/port/UserInterface/PythonPlayer.cpp @@ -1037,6 +1037,25 @@ void CPythonPlayer::SetSkillLevel_(DWORD dwSkillIndex, DWORD dwSkillGrade, DWORD m_playerStatus.aSkill[dwSlotIndex].fcurEfficientPercentage = LocaleService_GetSkillPower(dwSkillLevel)/100.0f; m_playerStatus.aSkill[dwSlotIndex].fnextEfficientPercentage = LocaleService_GetSkillPower(dwSkillLevel+1)/100.0f; + CInstanceBase * pkInstMain = NEW_GetMainActorPtr(); + if (pkInstMain) + { + if (dwSkillIndex == 4 && pkInstMain->IsAffect(CInstanceBase::AFFECT_GEOMGYEONG)) + { + pkInstMain->__Warrior_SetGeomgyeongAffect(true); + } + else + { + for (std::map::const_iterator it = m_kMap_dwAffectIndexToSkillIndex.begin(); it != m_kMap_dwAffectIndexToSkillIndex.end(); ++it) + { + if (it->second == dwSkillIndex && pkInstMain->IsAffect(it->first)) + { + pkInstMain->SCRIPT_SetAffect(it->first, false); + pkInstMain->SCRIPT_SetAffect(it->first, true); + } + } + } + } } void CPythonPlayer::SetSkillCoolTime(DWORD dwSkillIndex) diff --git a/extension/src/port/UserInterface/PythonSkill.cpp b/extension/src/port/UserInterface/PythonSkill.cpp index d52977f7..9e2d6f60 100644 --- a/extension/src/port/UserInterface/PythonSkill.cpp +++ b/extension/src/port/UserInterface/PythonSkill.cpp @@ -12,7 +12,7 @@ std::map CPythonSkill::SSkillData::ms_NewMinStatusNameMap; std::map CPythonSkill::SSkillData::ms_NewMaxStatusNameMap; DWORD CPythonSkill::SSkillData::ms_dwTimeIncreaseSkillNumber = 0; -BOOL SKILL_EFFECT_UPGRADE_ENABLE = FALSE; +BOOL SKILL_EFFECT_UPGRADE_ENABLE = TRUE; int SplitLine(const char * c_szText, CTokenVector* pstTokenVector, const char * c_szDelimeter) { diff --git a/extension/src/port/common/D3D8Types.h b/extension/src/port/common/D3D8Types.h index 1dda1824..93fd11b7 100644 --- a/extension/src/port/common/D3D8Types.h +++ b/extension/src/port/common/D3D8Types.h @@ -757,6 +757,7 @@ struct IDirect3DDevice8 virtual ULONG Release() = 0; virtual HRESULT GetDeviceCaps(D3DCAPS8* pCaps) = 0; virtual HRESULT GetViewport(D3DVIEWPORT8* pViewport) = 0; + virtual HRESULT SetViewport(const D3DVIEWPORT8* pViewport) = 0; virtual UINT GetAvailableTextureMem() = 0; virtual HRESULT BeginScene() = 0; virtual HRESULT EndScene() = 0; diff --git a/extension/src/python_host_node.cpp b/extension/src/python_host_node.cpp index 797170f3..95b15817 100644 --- a/extension/src/python_host_node.cpp +++ b/extension/src/python_host_node.cpp @@ -12,7 +12,13 @@ #include "platform/EterLib/UIRenderCommands.h" #include "platform/EterLib/RenderCommands3D.h" #include "platform/MilesLib/AudioCommands.h" +#include "platform/PackBackend.h" +#include "../../native_render/draw_capture.h" +#include +#include #include +#include +#include #endif #include @@ -23,6 +29,8 @@ #include #include #include +#include + using namespace godot; @@ -82,53 +90,188 @@ void Metin2PythonHost::ui_mouse_move(int x, int y) { PythonBoot::UIMouseMove(x, void Metin2PythonHost::ui_mouse_button(int button, bool pressed, int x, int y) { PythonBoot::UIMouseButton(button, pressed, x, y); } +void Metin2PythonHost::ui_mouse_wheel(int delta) { PythonBoot::UIMouseWheel(delta); } void Metin2PythonHost::ui_key(int key, bool pressed) { PythonBoot::UIKey(key, pressed); } void Metin2PythonHost::ui_char(int codepoint) { PythonBoot::UIChar(unsigned(codepoint)); } void Metin2PythonHost::ui_ime_key(int vkey) { PythonBoot::UIIMEKeyDown(vkey); } void Metin2PythonHost::ui_update() { PythonBoot::UIUpdate(); } +namespace { +Dictionary ui_command_item(const UIRenderCommand &command) { + static const StringName s_kind("kind"); + static const StringName s_x1("x1"); + static const StringName s_y1("y1"); + static const StringName s_x2("x2"); + static const StringName s_y2("y2"); + static const StringName s_argb("argb"); + static const StringName s_end_argb("end_argb"); + static const StringName s_clip_x1("clip_x1"); + static const StringName s_clip_y1("clip_y1"); + static const StringName s_clip_x2("clip_x2"); + static const StringName s_clip_y2("clip_y2"); + static const StringName s_text("text"); + static const StringName s_quad("quad"); + static const StringName s_uv("uv"); + static const StringName s_blend("blend"); + static const StringName s_mask("mask"); + static const StringName s_mask_uv("mask_uv"); + static const StringName s_behind_3d("behind_3d"); + + static const StringName s_val_bar("bar"); + static const StringName s_val_gradient_bar("gradient_bar"); + static const StringName s_val_line("line"); + static const StringName s_val_image("image"); + static const StringName s_val_text("text"); + Dictionary item; + item[s_kind] = command.kind == UIRenderCommand::GradientBar ? s_val_gradient_bar : + (command.kind == UIRenderCommand::Bar ? s_val_bar : + (command.kind == UIRenderCommand::Line ? s_val_line : + (command.kind == UIRenderCommand::Image ? s_val_image : s_val_text))); + item[s_x1] = command.x1; + item[s_y1] = command.y1; + item[s_x2] = command.x2; + item[s_y2] = command.y2; + item[s_argb] = static_cast(command.argb); + if (command.kind == UIRenderCommand::GradientBar) + item[s_end_argb] = static_cast(command.end_argb); + item[s_clip_x1] = command.clip_x1; + item[s_clip_y1] = command.clip_y1; + item[s_clip_x2] = command.clip_x2; + item[s_clip_y2] = command.clip_y2; + if (command.kind == UIRenderCommand::Text || command.kind == UIRenderCommand::Image) + item[s_text] = String::utf8(command.text.c_str()); + if (command.quad) { + PackedVector2Array quad; + quad.resize(4); + Vector2 *qptr = quad.ptrw(); + for (int i = 0; i < 4; ++i) + qptr[i] = Vector2(command.qx[i], command.qy[i]); + item[s_quad] = quad; + + PackedVector2Array uv; + uv.resize(4); + Vector2 *uvptr = uv.ptrw(); + uvptr[0] = Vector2(command.su, command.sv); + uvptr[1] = Vector2(command.eu, command.sv); + uvptr[2] = Vector2(command.su, command.ev); + uvptr[3] = Vector2(command.eu, command.ev); + item[s_uv] = uv; + item[s_blend] = command.blend; + if (!command.mask.empty()) { + item[s_mask] = String::utf8(command.mask.c_str()); + PackedVector2Array mask_uv; + mask_uv.resize(4); + Vector2 *mptr = mask_uv.ptrw(); + for (int i = 0; i < 4; ++i) + mptr[i] = Vector2(command.mu[i], command.mv[i]); + item[s_mask_uv] = mask_uv; + } + } + item[s_behind_3d] = command.behind_3d; + return item; +} +} // namespace + Array Metin2PythonHost::ui_render_commands() { PythonBoot::UIRender(); + static std::uint64_t cached_frame_id = UINT64_MAX; + static Array cached_commands; + const std::uint64_t frame_id = UIRenderFrameId(); + if (cached_frame_id == frame_id) + return cached_commands; + const auto &commands = UIRenderCommands(); Array out; - for (const auto &command : UIRenderCommands()) { - Dictionary item; - item["kind"] = command.kind == UIRenderCommand::GradientBar ? "gradient_bar" : - (command.kind == UIRenderCommand::Bar ? "bar" : - (command.kind == UIRenderCommand::Line ? "line" : - (command.kind == UIRenderCommand::Image ? "image" : "text"))); - item["x1"] = command.x1; - item["y1"] = command.y1; - item["x2"] = command.x2; - item["y2"] = command.y2; - item["argb"] = static_cast(command.argb); - if (command.kind == UIRenderCommand::GradientBar) - item["end_argb"] = static_cast(command.end_argb); - item["clip_x1"] = command.clip_x1; - item["clip_y1"] = command.clip_y1; - item["clip_x2"] = command.clip_x2; - item["clip_y2"] = command.clip_y2; - if (command.kind == UIRenderCommand::Text || command.kind == UIRenderCommand::Image) - item["text"] = String::utf8(command.text.c_str()); - if (command.quad) { - PackedVector2Array quad; - for (int i = 0; i < 4; ++i) - quad.push_back(Vector2(command.qx[i], command.qy[i])); - item["quad"] = quad; - item["uv"] = PackedVector2Array({Vector2(command.su, command.sv), Vector2(command.eu, command.sv), - Vector2(command.su, command.ev), Vector2(command.eu, command.ev)}); - item["blend"] = command.blend; - if (!command.mask.empty()) { - item["mask"] = String::utf8(command.mask.c_str()); - PackedVector2Array mask_uv; - for (int i = 0; i < 4; ++i) - mask_uv.push_back(Vector2(command.mu[i], command.mv[i])); - item["mask_uv"] = mask_uv; - } - } - out.push_back(item); - } + out.resize(static_cast(commands.size())); + for (size_t idx = 0; idx < commands.size(); ++idx) + out[static_cast(idx)] = ui_command_item(commands[idx]); + cached_frame_id = frame_id; + cached_commands = out; return out; } +Array Metin2PythonHost::ui_render_commands_batched() { + PythonBoot::UIRender(); + static std::uint64_t cached_frame_id = UINT64_MAX; + static Array cached_commands; + const std::uint64_t frame_id = UIRenderFrameId(); + if (cached_frame_id == frame_id) + return cached_commands; + + const auto &commands = UIRenderCommands(); + const auto batchable = [](const UIRenderCommand &c) { + return c.kind == UIRenderCommand::Image && c.quad && c.text.rfind("mem:", 0) == 0 && + c.mask.empty() && c.blend == 0 && c.x1 >= c.clip_x1 && c.y1 >= c.clip_y1 && + c.x2 <= c.clip_x2 && c.y2 <= c.clip_y2; + }; + Array out; + for (size_t i = 0; i < commands.size();) { + const UIRenderCommand &first = commands[i]; + if (!batchable(first)) { + out.push_back(ui_command_item(first)); + ++i; + continue; + } + size_t end = i + 1; + while (end < commands.size() && batchable(commands[end]) && + commands[end].text == first.text && commands[end].behind_3d == first.behind_3d) + ++end; + if (end == i + 1) { + out.push_back(ui_command_item(first)); + i = end; + continue; + } + const int64_t count = static_cast(end - i); + PackedVector2Array points, uvs; + PackedColorArray colors; + PackedInt32Array indices; + points.resize(count * 4); + uvs.resize(count * 4); + colors.resize(count * 4); + indices.resize(count * 6); + Vector2 *point_data = points.ptrw(), *uv_data = uvs.ptrw(); + Color *color_data = colors.ptrw(); + int32_t *index_data = indices.ptrw(); + for (int64_t j = 0; j < count; ++j) { + const UIRenderCommand &c = commands[i + static_cast(j)]; + const int64_t v = j * 4, t = j * 6; + point_data[v] = Vector2(c.qx[0], c.qy[0]); + point_data[v + 1] = Vector2(c.qx[1], c.qy[1]); + point_data[v + 2] = Vector2(c.qx[3], c.qy[3]); + point_data[v + 3] = Vector2(c.qx[2], c.qy[2]); + uv_data[v] = Vector2(c.su, c.sv); + uv_data[v + 1] = Vector2(c.eu, c.sv); + uv_data[v + 2] = Vector2(c.eu, c.ev); + uv_data[v + 3] = Vector2(c.su, c.ev); + const Color color(((c.argb >> 16) & 255) / 255.0f, ((c.argb >> 8) & 255) / 255.0f, + (c.argb & 255) / 255.0f, ((c.argb >> 24) & 255) / 255.0f); + for (int k = 0; k < 4; ++k) + color_data[v + k] = color; + const int32_t base = static_cast(v); + index_data[t] = base; + index_data[t + 1] = base + 1; + index_data[t + 2] = base + 2; + index_data[t + 3] = base; + index_data[t + 4] = base + 2; + index_data[t + 5] = base + 3; + } + Dictionary item; + item["kind"] = StringName("glyph_batch"); + item["text"] = String::utf8(first.text.c_str()); + item["behind_3d"] = first.behind_3d; + item["points"] = points; + item["uvs"] = uvs; + item["colors"] = colors; + item["indices"] = indices; + item["glyph_count"] = count; + out.push_back(item); + i = end; + } + cached_frame_id = frame_id; + cached_commands = out; + return out; +} + +bool Metin2PythonHost::has_3d_draws() { return !Render3DDraws().empty(); } + Ref Metin2PythonHost::memory_texture(const String &name, int64_t known_revision) { UIMemoryTexture texture; if (!UIRenderMemoryTexture(name.utf8().get_data(), &texture) || texture.width <= 0 || texture.height <= 0) @@ -162,12 +305,107 @@ Color argb_color(std::uint32_t argb) { ((argb >> 24) & 255) / 255.0f); } +bool read_pack_texture(const std::string &vpath, std::vector &bytes) { + if (vpath.empty() || !mtpack40250::ready()) + return false; + std::string norm = vpath; + for (char &ch : norm) + if (ch == '\\') + ch = '/'; + std::string stripped = norm; + if (stripped.size() >= 2 && stripped[1] == ':') + stripped = stripped.substr(2); + while (!stripped.empty() && stripped.front() == '/') + stripped.erase(stripped.begin()); + auto lower = [](std::string s) { + for (char &ch : s) + if (ch >= 'A' && ch <= 'Z') + ch = static_cast(ch - 'A' + 'a'); + return s; + }; + for (const std::string &candidate : { + norm, stripped, "d:/" + stripped, + lower(norm), lower(stripped), lower("d:/" + stripped)}) { + if (mtpack40250::read(candidate, bytes) && !bytes.empty()) + return true; + } + return false; +} + } // namespace Array Metin2PythonHost::render3d_draws() { + struct CachedGeometry { + std::uint64_t revision; + std::uint64_t last_used; + Dictionary arrays; + }; + static std::unordered_map geometry_cache; + static std::uint64_t extraction = 0; + ++extraction; + const auto &draws = Render3DDraws(); + static bool capture_written = false; + if (!capture_written) { + const char *capture_path = std::getenv("MT_NATIVE_CAPTURE_PATH"); + const char *minimum_text = std::getenv("MT_NATIVE_CAPTURE_MIN_DRAWS"); + const unsigned long minimum = minimum_text ? std::strtoul(minimum_text, nullptr, 10) : 100UL; + if (capture_path && *capture_path && draws.size() >= minimum) { + capture_written = true; + try { + std::unordered_map> textures; + auto add_texture = [&](const std::string &name) { + if (name.empty() || textures.count(name)) + return; + if (name.rfind("mem:", 0) == 0) { + UIMemoryTexture mem_tex; + if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { + auto mtra = native_draw_capture::encode_raw_argb_as_mtra( + static_cast(mem_tex.width), + static_cast(mem_tex.height), + mem_tex.argb.data()); + if (!mtra.empty()) + textures.emplace(name, std::move(mtra)); + } + return; + } + std::vector bytes; + if (read_pack_texture(name, bytes)) + textures.emplace(name, std::move(bytes)); + }; + for (const Render3DDraw &draw : draws) { + add_texture(draw.texture0); + add_texture(draw.texture1); + } + const auto &ui_commands = UIRenderCommands(); + for (const UIRenderCommand &cmd : ui_commands) { + if (cmd.kind == UIRenderCommand::Image) { + add_texture(cmd.text); + add_texture(cmd.mask); + } + } + unsigned ui_w = 0, ui_h = 0; + UIRenderGetSize(&ui_w, &ui_h); + native_draw_capture::write( + capture_path, draws, textures, + ui_w ? ui_w : 960u, ui_h ? ui_h : 640u, ui_commands); + UtilityFunctions::print( + "native draw capture: ", capture_path, + " draws=", static_cast(draws.size()), + " ui_commands=", static_cast(ui_commands.size()), + " textures=", static_cast(textures.size())); + } catch (const std::exception &error) { + UtilityFunctions::printerr("native draw capture failed: ", error.what()); + } + } + } + Array out; - for (const Render3DDraw &draw : Render3DDraws()) { + out.resize(static_cast(draws.size())); + int64_t draw_index = 0; + for (const Render3DDraw &draw : draws) { Dictionary item; + item["geometry_key"] = static_cast(draw.geometry_key); + item["geometry_revision"] = static_cast(draw.geometry_revision); item["world"] = floats(draw.world, 16); item["view"] = floats(draw.view, 16); item["proj"] = floats(draw.proj, 16); @@ -176,43 +414,73 @@ Array Metin2PythonHost::render3d_draws() { item["pretransformed"] = draw.pretransformed; item["lines"] = draw.lines; - PackedVector3Array positions; - positions.resize(static_cast(draw.positions.size() / 3)); - for (int64_t i = 0; i < positions.size(); ++i) - positions[i] = Vector3(draw.positions[i * 3], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2]); - item["positions"] = positions; - if (!draw.rhw.empty()) - item["rhw"] = floats(draw.rhw.data(), static_cast(draw.rhw.size())); - if (!draw.normals.empty()) { - PackedVector3Array normals; - normals.resize(static_cast(draw.normals.size() / 3)); - for (int64_t i = 0; i < normals.size(); ++i) - normals[i] = Vector3(draw.normals[i * 3], draw.normals[i * 3 + 1], draw.normals[i * 3 + 2]); - item["normals"] = normals; + Dictionary geometry; + if (draw.geometry_key != 0) { + auto cached = geometry_cache.find(draw.geometry_key); + if (cached != geometry_cache.end() && cached->second.revision == draw.geometry_revision) { + cached->second.last_used = extraction; + geometry = cached->second.arrays; + } } - auto uvs = [](const std::vector &values) { + if (geometry.is_empty()) { + PackedVector3Array positions; + positions.resize(static_cast(draw.positions.size() / 3)); + if constexpr (sizeof(Vector3) == sizeof(float) * 3 && std::is_trivially_copyable_v) { + if (!draw.positions.empty()) + std::memcpy(positions.ptrw(), draw.positions.data(), draw.positions.size() * sizeof(float)); + } else { + Vector3 *dst = positions.ptrw(); + for (int64_t i = 0; i < positions.size(); ++i) + dst[i] = Vector3(draw.positions[i * 3], draw.positions[i * 3 + 1], draw.positions[i * 3 + 2]); + } + geometry["positions"] = positions; + if (!draw.rhw.empty()) + geometry["rhw"] = floats(draw.rhw.data(), static_cast(draw.rhw.size())); + if (!draw.normals.empty()) { + PackedVector3Array normals; + normals.resize(static_cast(draw.normals.size() / 3)); + if constexpr (sizeof(Vector3) == sizeof(float) * 3 && std::is_trivially_copyable_v) { + std::memcpy(normals.ptrw(), draw.normals.data(), draw.normals.size() * sizeof(float)); + } else { + Vector3 *dst = normals.ptrw(); + for (int64_t i = 0; i < normals.size(); ++i) + dst[i] = Vector3(draw.normals[i * 3], draw.normals[i * 3 + 1], draw.normals[i * 3 + 2]); + } + geometry["normals"] = normals; + } + auto uvs = [](const std::vector &values) { PackedVector2Array out; out.resize(static_cast(values.size() / 2)); - for (int64_t i = 0; i < out.size(); ++i) - out[i] = Vector2(values[i * 2], values[i * 2 + 1]); + if constexpr (sizeof(Vector2) == sizeof(float) * 2 && std::is_trivially_copyable_v) { + if (!values.empty()) + std::memcpy(out.ptrw(), values.data(), values.size() * sizeof(float)); + } else { + Vector2 *dst = out.ptrw(); + for (int64_t i = 0; i < out.size(); ++i) + dst[i] = Vector2(values[i * 2], values[i * 2 + 1]); + } return out; }; - if (!draw.uv0.empty()) - item["uv0"] = uvs(draw.uv0); - if (!draw.uv1.empty()) - item["uv1"] = uvs(draw.uv1); - if (!draw.diffuse.empty()) { - PackedColorArray colors; - colors.resize(static_cast(draw.diffuse.size())); - for (int64_t i = 0; i < colors.size(); ++i) - colors[i] = argb_color(draw.diffuse[i]); - item["diffuse"] = colors; + if (!draw.uv0.empty()) + geometry["uv0"] = uvs(draw.uv0); + if (!draw.uv1.empty()) + geometry["uv1"] = uvs(draw.uv1); + if (!draw.diffuse.empty()) { + PackedColorArray colors; + colors.resize(static_cast(draw.diffuse.size())); + for (int64_t i = 0; i < colors.size(); ++i) + colors[i] = argb_color(draw.diffuse[i]); + geometry["diffuse"] = colors; + } + PackedInt32Array indices; + indices.resize(static_cast(draw.indices.size())); + for (int64_t i = 0; i < indices.size(); ++i) + indices[i] = static_cast(draw.indices[i]); + geometry["indices"] = indices; + if (draw.geometry_key != 0) + geometry_cache[draw.geometry_key] = {draw.geometry_revision, extraction, geometry}; } - PackedInt32Array indices; - indices.resize(static_cast(draw.indices.size())); - for (int64_t i = 0; i < indices.size(); ++i) - indices[i] = static_cast(draw.indices[i]); - item["indices"] = indices; + item.merge(geometry); item["alpha_blend"] = static_cast(draw.alpha_blend); item["src_blend"] = static_cast(draw.src_blend); @@ -226,6 +494,16 @@ Array Metin2PythonHost::render3d_draws() { item["z_func"] = static_cast(draw.z_func); item["lighting"] = static_cast(draw.lighting); item["texture_factor"] = static_cast(draw.texture_factor); + bool uses_tf = false; + for (int s = 0; s < 2; ++s) { + if (draw.color_op[s] > 1 && + (((draw.color_arg1[s] & 0xF) == 3) || ((draw.color_arg2[s] & 0xF) == 3))) + uses_tf = true; + if (draw.alpha_op[s] > 1 && + (((draw.alpha_arg1[s] & 0xF) == 3) || ((draw.alpha_arg2[s] & 0xF) == 3))) + uses_tf = true; + } + item["uses_tf"] = uses_tf; item["fog_enable"] = static_cast(draw.fog_enable); item["color_op"] = static_cast(draw.color_op[0]); item["alpha_op"] = static_cast(draw.alpha_op[0]); @@ -243,7 +521,22 @@ Array Metin2PythonHost::render3d_draws() { item["light0_ambient"] = Color(draw.light0_ambient[0], draw.light0_ambient[1], draw.light0_ambient[2], draw.light0_ambient[3]); item["ambient"] = argb_color(draw.ambient); - out.push_back(item); + PackedFloat32Array vp; + vp.resize(4); + vp[0] = draw.viewport[0]; + vp[1] = draw.viewport[1]; + vp[2] = draw.viewport[2]; + vp[3] = draw.viewport[3]; + item["viewport"] = vp; + out[draw_index++] = item; + } + if (extraction % 120 == 0) { + for (auto it = geometry_cache.begin(); it != geometry_cache.end();) { + if (extraction - it->second.last_used > 120) + it = geometry_cache.erase(it); + else + ++it; + } } return out; } @@ -314,11 +607,14 @@ bool Metin2PythonHost::is_app_looping() { return false; } void Metin2PythonHost::set_ui_size(int, int) {} void Metin2PythonHost::ui_mouse_move(int, int) {} void Metin2PythonHost::ui_mouse_button(int, bool, int, int) {} +void Metin2PythonHost::ui_mouse_wheel(int) {} void Metin2PythonHost::ui_key(int, bool) {} void Metin2PythonHost::ui_char(int) {} void Metin2PythonHost::ui_ime_key(int) {} void Metin2PythonHost::ui_update() {} Array Metin2PythonHost::ui_render_commands() { return Array(); } +Array Metin2PythonHost::ui_render_commands_batched() { return Array(); } +bool Metin2PythonHost::has_3d_draws() { return false; } Ref Metin2PythonHost::memory_texture(const String &, int64_t) { return Ref(); } Array Metin2PythonHost::render3d_draws() { return Array(); } Array Metin2PythonHost::audio_commands() { return Array(); } @@ -341,11 +637,14 @@ void Metin2PythonHost::_bind_methods() { ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("set_ui_size", "width", "height"), &Metin2PythonHost::set_ui_size); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_mouse_move", "x", "y"), &Metin2PythonHost::ui_mouse_move); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_mouse_button", "button", "pressed", "x", "y"), &Metin2PythonHost::ui_mouse_button); + ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_mouse_wheel", "delta"), &Metin2PythonHost::ui_mouse_wheel); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_key", "key", "pressed"), &Metin2PythonHost::ui_key); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_char", "codepoint"), &Metin2PythonHost::ui_char); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_ime_key", "vkey"), &Metin2PythonHost::ui_ime_key); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_update"), &Metin2PythonHost::ui_update); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_render_commands"), &Metin2PythonHost::ui_render_commands); + ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("ui_render_commands_batched"), &Metin2PythonHost::ui_render_commands_batched); + ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("has_3d_draws"), &Metin2PythonHost::has_3d_draws); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("memory_texture", "name", "known_revision"), &Metin2PythonHost::memory_texture, DEFVAL(-1)); ClassDB::bind_static_method("Metin2PythonHost", D_METHOD("render3d_draws"), &Metin2PythonHost::render3d_draws); diff --git a/extension/src/python_host_node.h b/extension/src/python_host_node.h index d77b4719..43eeda6e 100644 --- a/extension/src/python_host_node.h +++ b/extension/src/python_host_node.h @@ -38,12 +38,16 @@ public: static void set_ui_size(int width, int height); static void ui_mouse_move(int x, int y); static void ui_mouse_button(int button, bool pressed, int x, int y); + static void ui_mouse_wheel(int delta); static void ui_key(int key, bool pressed); // WM_CHAR (Unicode code point) and WM_KEYDOWN (Win32 VK code) for the 40250 IME. static void ui_char(int codepoint); static void ui_ime_key(int vkey); static void ui_update(); static godot::Array ui_render_commands(); + // Ordered UI commands with consecutive, unclipped font quads packed into triangle arrays. + static godot::Array ui_render_commands_batched(); + static bool has_3d_draws(); // The pixels of a "mem:@" image command (the CGraphicFontTexture glyph pages) as an // RGBA8 Image; null when the texture is gone or its revision is still `known_revision`. static godot::Ref memory_texture(const godot::String &name, int64_t known_revision); diff --git a/extension/src/terrain_splat.cpp b/extension/src/terrain_splat.cpp index 75042209..5733220e 100644 --- a/extension/src/terrain_splat.cpp +++ b/extension/src/terrain_splat.cpp @@ -14,7 +14,11 @@ #include #include +#include +#include +#include #include +#include using namespace godot; @@ -72,14 +76,23 @@ Ref terrain_shader() { // DDS -> RGBA8 Image,resize 到 size×size。缓存(非函数静态 —— 见 cleanup)。 std::unordered_map> g_layer_cache; -Ref layer_image(const std::string &real_path, int size) { +std::unordered_map> g_layer_dimensions; +Ref layer_image(const std::string &real_path, int size, + std::unordered_map &decoded_sources) { auto &cache = g_layer_cache; std::string key = real_path + "@" + std::to_string(size); auto it = cache.find(key); if (it != cache.end()) return it->second; Ref out; - mtgodot::Image d = mtgodot::dds_from_file(godot::String(real_path.c_str())); + mtgodot::Image d; + auto decoded = decoded_sources.find(real_path); + if (decoded != decoded_sources.end()) { + d = std::move(decoded->second); + decoded_sources.erase(decoded); + } else { + d = mtgodot::dds_from_file(godot::String(real_path.c_str())); + } if (d.ok()) { PackedByteArray b; b.resize((int64_t)d.rgba.size()); @@ -120,25 +133,38 @@ Ref weight_tex(const std::vector &four) { void cleanup_terrain_shader() { g_terrain_shader.unref(); g_layer_cache.clear(); // 释放缓存的 Ref,别拖到 __cxa_finalize(那时引擎已析构) + g_layer_dimensions.clear(); } Ref build_chunk_terrain_material(const fmt::SplatSet &splat, const fmt::TextureSet &tset, const fmt::AssetResolver &res, const String &shadowmap_path) { + const bool profile = std::getenv("MT_PROFILE_MAP") != nullptr; + const auto now = [] { return std::chrono::steady_clock::now(); }; + const auto start = now(); // Texture2DArray 要求各 slice 同尺寸 —— 取用到的图层里的最大源边长(上限 1024), // 只放大不缩小最大源,避免把 512² 地表贴图硬降采样(PARITY-GAP §3.4)。 int src_max = 256; + // Keep newly decoded sources only until the image pass consumes them. The size pass and image pass + // used to decode the same DDS twice on its first use. + std::unordered_map decoded_sources; for (const auto &L : splat.layers) { if (L.layer >= 1 && L.layer <= (int)tset.layers.size()) { std::string rp = res.resolve(tset.layers[L.layer - 1].texture, nullptr); if (rp.empty()) continue; - mtgodot::Image d = mtgodot::dds_from_file(godot::String(rp.c_str())); - if (d.ok()) - src_max = std::max(src_max, std::max(d.w, d.h)); + auto it = g_layer_dimensions.find(rp); + if (it == g_layer_dimensions.end()) { + mtgodot::Image d = mtgodot::dds_from_file(godot::String(rp.c_str())); + it = g_layer_dimensions.emplace(rp, d.ok() ? std::make_pair(int(d.w), int(d.h)) + : std::make_pair(0, 0)).first; + decoded_sources.emplace(rp, std::move(d)); + } + src_max = std::max(src_max, std::max(it->second.first, it->second.second)); } } const int LSIZE = std::min(1024, src_max); + const auto dimensions_done = now(); int n = std::min(MAX_LAYERS, (int)splat.layers.size()); if (n == 0) @@ -167,7 +193,7 @@ Ref build_chunk_terrain_material(const fmt::SplatSet &splat, const fmt::TextureLayer &tl = tset.layers[L.layer - 1]; std::string rp = res.resolve(tl.texture, nullptr); if (!rp.empty()) - img = layer_image(rp, LSIZE); + img = layer_image(rp, LSIZE, decoded_sources); // 原客户端 TextureSet.cpp:185:u' = (TexCoordBase*UScale)*vtx_cm + UOffset, // TexCoordBase = 1/(PATCH_XSIZE*CELLSCALE) = 1/3200;区块归一化 UV -> 平铺频率 = 8*Scale。 float us = tl.u_scale > 0.01f ? tl.u_scale : 1.0f; @@ -176,10 +202,12 @@ Ref build_chunk_terrain_material(const fmt::SplatSet &splat, } imgs.push_back(img.is_valid() ? img : fallback); } + const auto images_done = now(); Ref arr; arr.instantiate(); arr->create_from_images(imgs); + const auto array_done = now(); // 权重贴图:ceil(n/4) 张 RGBA8(每通道一层 alpha) Ref wtex[4]; @@ -192,6 +220,7 @@ Ref build_chunk_terrain_material(const fmt::SplatSet &splat, } wtex[g] = weight_tex(grp); } + const auto weights_done = now(); Ref mat; mat.instantiate(); @@ -209,6 +238,7 @@ Ref build_chunk_terrain_material(const fmt::SplatSet &splat, mat->set_shader_parameter("layer_uv", uva); } + const auto shader_done = now(); if (!shadowmap_path.is_empty()) { mtgodot::Image sm = mtgodot::dds_from_file(shadowmap_path); if (sm.ok()) { @@ -221,6 +251,13 @@ Ref build_chunk_terrain_material(const fmt::SplatSet &splat, mat->set_shader_parameter("use_shadowmap", true); } } + if (profile) { + const auto ms = [](auto a, auto b) { return std::chrono::duration(b - a).count(); }; + std::fprintf(stderr, "MATERIAL_PROFILE dimensions=%.3f images=%.3f array=%.3f weights=%.3f shader=%.3f shadow=%.3f total=%.3f layers=%d size=%d\n", + ms(start, dimensions_done), ms(dimensions_done, images_done), ms(images_done, array_done), + ms(array_done, weights_done), ms(weights_done, shader_done), ms(shader_done, now()), + ms(start, now()), n, LSIZE); + } return mat; } diff --git a/extension/tests/port_login_flow_server.cpp b/extension/tests/port_login_flow_server.cpp index babbd75b..8a9f9623 100644 --- a/extension/tests/port_login_flow_server.cpp +++ b/extension/tests/port_login_flow_server.cpp @@ -13,6 +13,7 @@ #include #include +#include #include using namespace mtnet::classic; @@ -23,6 +24,16 @@ constexpr std::uint32_t kHandshake = 0x2468ace0; constexpr int kHandshakeRetryLimit = 32; // game/src/desc.h HANDSHAKE_RETRY_LIMIT constexpr int kIdleMs = 20000; +int fake_mob_count() +{ + const char* value = std::getenv("MT_FAKE_MOB_COUNT"); + if (!value || !*value) + return 1; + char* end = nullptr; + const long count = std::strtol(value, &end, 10); + return *end == '\0' && count >= 1 && count <= 64 ? static_cast(count) : 1; +} + // get_dword_time(): milliseconds on the server's clock. std::uint32_t now_ms() { @@ -582,6 +593,20 @@ bool FakeLoginServer::ServeGame(Connection& c, int index) mob_update.attack_speed = 100; if (!c.Send(mob) || !c.Send(mob_update)) return Fail(name + ": send monster"); + // Optional crowd for render profiling. Keep the primary dog and its click path clear so the + // existing combat test still attacks kMobVID; additional dogs are passive scenery. + const int mob_count = fake_mob_count(); + for (int i = 1; i < mob_count; ++i) + { + GCCharacterAdd extra = mob; + extra.vid = kMobVID + static_cast(i); + extra.x = kMobX - 350 - ((i - 1) % 8) * 170; + extra.y = kMobY + 250 + ((i - 1) / 8) * 190; + GCCharacterUpdate extra_update = mob_update; + extra_update.vid = extra.vid; + if (!c.Send(extra) || !c.Send(extra_update)) + return Fail(name + ": send extra monster"); + } // An NPC is ADD + ADDITIONAL_INFO + UPDATE like a PC (the name is the mob_proto locale name). GCCharacterAdd keeper = zeroed(); keeper.header = HDR_GC_CHARACTER_ADD; diff --git a/extension/tests/port_login_flow_test.cpp b/extension/tests/port_login_flow_test.cpp index 8e49ac2d..7826d806 100644 --- a/extension/tests/port_login_flow_test.cpp +++ b/extension/tests/port_login_flow_test.cpp @@ -49,10 +49,12 @@ #include "GameLib/FlyingData.h" #include "GameLib/FlyingObjectManager.h" #include "EffectLib/EffectManager.h" +#include "EffectLib/EffectInstance.h" #include "EterLib/Camera.h" #include "GameLib/ItemManager.h" #include "UserInterface/PythonItem.h" #include "UserInterface/StdAfx.h" +#include "UserInterface/PythonPlayer.h" #include "UserInterface/PythonExchange.h" #include "UserInterface/PythonTextTail.h" #include "UserInterface/PythonCharacterManager.h" @@ -674,11 +676,11 @@ int main(int argc, char** argv) const size_t moves_before = server ? server->Events().size() : 0; if (main_instance) main_instance->NEW_GetPixelPosition(&start); - PythonBoot::UIMouseButton(1, true, 400, 520); + PythonBoot::UIMouseButton(1, true, 400, 220); pump_until(0.1, [] { return false; }); - PythonBoot::UIMouseButton(1, false, 400, 520); + PythonBoot::UIMouseButton(1, false, 400, 220); CHECK(main_instance && pump_until(3, [&] { return walked() > 100.0f; })); - CHECK(main_instance && pump_until(5, [&] { return !main_instance->IsWalking(); })); + CHECK(main_instance && pump_until(10, [&] { return !main_instance->IsWalking(); })); std::printf("port_login_flow_test: click walked %.0f cm\n", main_instance ? walked() : 0.0f); if (server) { @@ -746,6 +748,61 @@ int main(int argc, char** argv) ++attacks; std::printf("port_login_flow_test: %zu attack(s) until the stray dog died\n", attacks); CHECK(attacks >= (size_t) FakeLoginServer::kMobHits); + + // Camera mouse wheel zoom test: + // Test zooming in (Wheel UP, positive delta) and zooming out (Wheel DOWN, negative delta). + { + CCamera* pkCmrCur = CCameraManager::Instance().GetCurrentCamera(); + CHECK(pkCmrCur != nullptr); + if (pkCmrCur) + { + const float initial_dist = pkCmrCur->GetDistance(); + std::printf("port_login_flow_test: initial camera distance = %f\n", initial_dist); + + // Zoom IN: positive wheel delta + PythonBoot::UIMouseWheel(120); + pump_until(0.1, [] { return false; }); + const float zoomed_in_dist = pkCmrCur->GetDistance(); + std::printf("port_login_flow_test: zoomed in camera distance = %f\n", zoomed_in_dist); + CHECK(zoomed_in_dist < initial_dist); + + // Zoom OUT: negative wheel delta + PythonBoot::UIMouseWheel(-240); + pump_until(0.1, [] { return false; }); + const float zoomed_out_dist = pkCmrCur->GetDistance(); + std::printf("port_login_flow_test: zoomed out camera distance = %f\n", zoomed_out_dist); + CHECK(zoomed_out_dist > zoomed_in_dist); + } + } + + // P-grade skill test: verify P-grade skill (geomgyeong / Aura of the Sword) uses Grade 3 motion (geomgyeong_4.msa), + // spawns effect instances (geom_4_badak.mse, geom_4_sword_making.mse), and renders them. + { + CPythonPlayer& rkPlayer = CPythonPlayer::Instance(); + rkPlayer.SetStatus(POINT_SP, 1000); + rkPlayer.SetStatus(POINT_MAX_SP, 1000); + rkPlayer.SetStatus(POINT_HP, 1000); + rkPlayer.SetStatus(POINT_MAX_HP, 1000); + CInstanceBase* pkInstMain = rkPlayer.NEW_GetMainActorPtr(); + CHECK(pkInstMain != nullptr); + if (pkInstMain) + { + pkInstMain->ChangeWeapon(19); + rkPlayer.SetSkill(4, 4); // slot 4 = geomgyeong + rkPlayer.SetSkillLevel_(4, 3, 40); // Grade 3 (P), Level 40 + CHECK(rkPlayer.GetSkillGrade(4) == 3); + rkPlayer.ClickSkillSlot(4); + CHECK(pkInstMain->IsUsingSkill()); + bool has_effects = false; + for (int step = 0; step < 10; ++step) + { + pump_until(0.1, [] { return false; }); + if (CEffectInstance::GetRenderingEffectCount() > 0) + has_effects = true; + } + CHECK(has_effects); + } + } } } // 10. (批次 4-c) NPC shop: a click on the general store walks up to it (__OnClickActor) and sends CG_ON_CLICK; diff --git a/native_render/CMakeLists.txt b/native_render/CMakeLists.txt new file mode 100644 index 00000000..d67acbb0 --- /dev/null +++ b/native_render/CMakeLists.txt @@ -0,0 +1,84 @@ +find_package(Vulkan REQUIRED) +find_package(SDL3 CONFIG REQUIRED) +find_program(GLSLC glslc REQUIRED) + +set(MT_NATIVE_VERT_SPV "${CMAKE_CURRENT_BINARY_DIR}/native.vert.spv") +set(MT_NATIVE_FRAG_SPV "${CMAKE_CURRENT_BINARY_DIR}/native.frag.spv") +add_custom_command(OUTPUT "${MT_NATIVE_VERT_SPV}" + COMMAND "${GLSLC}" -fshader-stage=vert "${CMAKE_CURRENT_SOURCE_DIR}/native.vert" -o "${MT_NATIVE_VERT_SPV}" + DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/native.vert") +add_custom_command(OUTPUT "${MT_NATIVE_FRAG_SPV}" + COMMAND "${GLSLC}" -fshader-stage=frag "${CMAKE_CURRENT_SOURCE_DIR}/native.frag" -o "${MT_NATIVE_FRAG_SPV}" + DEPENDS "${CMAKE_CURRENT_SOURCE_DIR}/native.frag") +add_custom_target(mt_native_shaders DEPENDS "${MT_NATIVE_VERT_SPV}" "${MT_NATIVE_FRAG_SPV}") + +set(MT_NATIVE_RENDER_SOURCES + main.cpp + stb_image_impl.cpp + "${PROJECT_SOURCE_DIR}/extension/src/dxt.cpp" +) +if(TARGET port_platform AND TARGET mtpython) + list(APPEND MT_NATIVE_RENDER_SOURCES + "${PROJECT_SOURCE_DIR}/extension/tests/port_login_flow_server.cpp" + "${PROJECT_SOURCE_DIR}/extension/src/net/classic/classic_cipher.cpp" + ) +endif() + +if(ANDROID) + # SDLActivity loads libmain.so; Android does not launch a native executable. + set(MT_NATIVE_TARGET main) + add_library(${MT_NATIVE_TARGET} SHARED ${MT_NATIVE_RENDER_SOURCES}) +else() + set(MT_NATIVE_TARGET mt_native_render) + add_executable(${MT_NATIVE_TARGET} ${MT_NATIVE_RENDER_SOURCES}) +endif() +if(APPLE) + set_target_properties(${MT_NATIVE_TARGET} PROPERTIES + OSX_ARCHITECTURES "${CMAKE_HOST_SYSTEM_PROCESSOR}" + ) +endif() +add_dependencies(${MT_NATIVE_TARGET} mt_native_shaders) +target_include_directories(${MT_NATIVE_TARGET} PRIVATE + "${CMAKE_CURRENT_SOURCE_DIR}/third_party" + "${PROJECT_SOURCE_DIR}/extension/src" + "${PROJECT_SOURCE_DIR}/extension/src/platform/EterLib") +if(NOT ANDROID) + add_custom_command(TARGET ${MT_NATIVE_TARGET} POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + "${MT_NATIVE_VERT_SPV}" "$/native.vert.spv" + COMMAND ${CMAKE_COMMAND} -E copy_if_different + "${MT_NATIVE_FRAG_SPV}" "$/native.frag.spv") +endif() +target_link_libraries(${MT_NATIVE_TARGET} PRIVATE Vulkan::Vulkan SDL3::SDL3) +if(APPLE) + target_link_libraries(${MT_NATIVE_TARGET} PRIVATE + "-framework AudioToolbox") +endif() +if(TARGET port_platform AND TARGET mtpython) + target_compile_definitions(${MT_NATIVE_TARGET} PRIVATE MT_NATIVE_HAS_LIVE_CLIENT=1) + target_link_libraries(${MT_NATIVE_TARGET} PRIVATE port_platform) + if(TARGET mtpython_stdlib) + add_dependencies(${MT_NATIVE_TARGET} mtpython_stdlib) + if(NOT ANDROID) + add_custom_command(TARGET ${MT_NATIVE_TARGET} POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + "${MT_PYTHON_STDLIB_ZIP}" "$/python27.zip") + endif() + endif() +endif() + +if(ANDROID) + # Add this directory to the SDL Android app's assets.srcDirs. + set(MT_NATIVE_ANDROID_ASSETS "${CMAKE_CURRENT_BINARY_DIR}/android-assets") + add_custom_target(mt_native_android_assets ALL + COMMAND ${CMAKE_COMMAND} -E make_directory "${MT_NATIVE_ANDROID_ASSETS}" + COMMAND ${CMAKE_COMMAND} -E copy_if_different "${MT_NATIVE_VERT_SPV}" "${MT_NATIVE_ANDROID_ASSETS}/native.vert.spv" + COMMAND ${CMAKE_COMMAND} -E copy_if_different "${MT_NATIVE_FRAG_SPV}" "${MT_NATIVE_ANDROID_ASSETS}/native.frag.spv" + DEPENDS mt_native_shaders) + if(TARGET mtpython_stdlib) + add_dependencies(mt_native_android_assets mtpython_stdlib) + add_custom_command(TARGET mt_native_android_assets POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + "${MT_PYTHON_STDLIB_ZIP}" "${MT_NATIVE_ANDROID_ASSETS}/python27.zip") + endif() +endif() diff --git a/native_render/README.md b/native_render/README.md new file mode 100644 index 00000000..15b59cd2 --- /dev/null +++ b/native_render/README.md @@ -0,0 +1,153 @@ +# Native Vulkan renderer prototype + +This is the standalone renderer path for the 40250 `Render3DDraw` and `UIRenderCommand` command streams, plus direct `--live-client` execution of the ported 40250 client (`PythonBoot`). SDL3 owns the window and input forwarding; Vulkan owns the swapchain (`FIFO` or `IMMEDIATE`/`MAILBOX`), `VK_FORMAT_D32_SFLOAT` depth buffer, `VK_QUERY_TYPE_TIMESTAMP` hardware GPU timer, key-indexed persistent vertex/index buffers, GPU skeletal skinning bone palette SSBO (`set = 1, binding = 0`), DDS/TGA/`mem:` glyph-page texture sampler descriptors, fixed-function 3D + 2D UI state pipelines, and draw submission. On macOS, the Vulkan loader uses MoltenVK over Metal. The existing Godot client remains the default playable path. + +## Build and run on macOS + +Install Vulkan headers/loader, MoltenVK, SDL3 and `glslc` (shaderc). With Homebrew: + +```sh +brew install vulkan-headers vulkan-loader molten-vk sdl3 shaderc + +# 1. Standalone replay build (without port_platform) +cmake -S . -B build-native-render -DMTGODOT_BUILD_EXTENSION=OFF \ + -DMT_BUILD_NATIVE_RENDER_PROTOTYPE=ON -DCMAKE_PREFIX_PATH=/opt/homebrew +cmake --build build-native-render --target mt_native_render -j8 + +# 2. Full live-client build (links port_platform + mtpython + FakeLoginServer) +cmake -S . -B build -DMT_BUILD_NATIVE_RENDER_PROTOTYPE=ON -DCMAKE_PREFIX_PATH=/opt/homebrew +cmake --build build --target mtgodot mt_native_render port_fake_login_server -j8 + +# 3. Optimized Release (-O3) live-client build +cmake -S . -B build-release -DCMAKE_BUILD_TYPE=Release -DMT_BUILD_NATIVE_RENDER_PROTOTYPE=ON \ + -DMTGODOT_EMBED_PYTHON=ON -DCMAKE_PREFIX_PATH=/opt/homebrew +cmake --build build-release --target mt_native_render -j8 +``` + +On macOS, `mt_native_render` automatically detects `/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json` (or `/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json`) when `VK_ICD_FILENAMES` is not set in the environment. +The build copies `native.vert.spv`, `native.frag.spv`, and, for the live client, +`python27.zip` beside the executable. Keep these files together when moving the +binary. In a macOS `.app`, put them in `Contents/Resources` (the directory +returned by SDL's `SDL_GetBasePath`). `MT_PYTHON_STDLIB` can override the zip path. + +## Android integration + +With an Android toolchain and SDL3 Android AAR/Prefab available to CMake, the +same option builds `libmain.so` for SDLActivity. The `mt_native_android_assets` +target prepares `native.vert.spv`, `native.frag.spv`, and `python27.zip` under +`/native_render/android-assets` (`python27.zip` is included when the +embedded Python target is enabled); include that directory in the SDL app's +`assets.srcDirs`. The app must allow network access for login. The 40250 `Client` +directory must be placed at `SDL_GetPrefPath("mtgodot", "native-render")/Client`, +with `pack/Index` present, before live mode starts. The Godot APK is a separate +application and does not launch this SDL renderer. + +The standalone `arm64-v8a` renderer cross-builds at Android API 24. The full +live-client cross-build currently stops in `extension/src/port/common/Win32Crt.cpp`: +the NDK exposes `` at API 24 but does not declare `iconv` until API 28. +This is a port-runtime prerequisite for an Android live-client APK; raising the +minimum API level is not assumed here. + +The Vulkan portability enumeration extension is selected only when advertised +by the loader. DDS, TGA, and memory textures retain their existing decoders; +JPEG, PNG, and BMP use the same portable decoder on macOS and Android. The +vendored `stb_image.h` is upstream v2.30 (SHA-256 +`594c2fe35d49488b4382dbfaec8f98366defca819d916ac95becf3e75f4200b3`). + +### Interactive Playable Modes + +Run the full 40250 client interactively (infinite frame loop until window close, resizable SDL3 window with automatic Vulkan swapchain recreation and `PythonBoot::SetUISize` sync, full keyboard/IME text input, SDL hardware cursor built from the original cursor images, and SDL3 + `AudioToolbox` `.wav`/`.mp3` audio): + +```sh +# Interactive outdoor map session (auto-login via loopback FakeLoginServer) +./build-release/native_render/mt_native_render \ + --live-client "/path/to/40250/Server Client TMP4/Client" \ + --interactive --fake-mobs 24 --width 1280 --height 800 + +# Interactive login screen (stops at introLogin.LoginWindow for manual typing/login) +./build-release/native_render/mt_native_render \ + --live-client "/path/to/40250/Server Client TMP4/Client" \ + --login-screen --width 1024 --height 768 + +# Connect to an external 40250 Auth + Game server +./build-release/native_render/mt_native_render \ + --live-client "/path/to/40250/Server Client TMP4/Client" \ + --live-server 127.0.0.1:11002:13000 --login-screen +``` + +### Synthetic benchmark + +```sh +./build-release/native_render/mt_native_render --frames 60 --draws 64 --triangles-per-draw 333 --no-vsync +``` + +Timing metrics printed by `mt_native_render`: +- `p95_frame_ms`, `p99_frame_ms`, `max_frame_ms`: wall-clock time for each measured update/render iteration, excluding startup and final GPU drain. These include vsync wait when enabled; use them alongside `gpu_ms` and the CPU breakdown. +- `game_update_ms`: mean CPU time spent in `PythonBoot::UIUpdate()`, `PythonBoot::UIRender()`, and audio command draining per frame in `--live-client` mode. +- `prepare_ms`: mean CPU draw-preparation time across all frames (including frame 0 cold-start geometry/texture uploads and pipeline creation). +- `steady_prepare_ms`: mean CPU draw-preparation time on frames after frame 0 (bone palette copy, UI quad batching, command recording). +- `submit_ms`: host CPU time around `vkQueueSubmit` (in `FIFO` mode this includes swapchain backpressure; pass `--no-vsync` to switch to `IMMEDIATE`/`MAILBOX`). +- `gpu_ms`: true hardware GPU execution time between top-of-pipe `vkCmdBeginRenderPass` and bottom-of-pipe `vkCmdEndRenderPass` measured via `VK_QUERY_TYPE_TIMESTAMP`. + +### macOS real-server acceptance + +Build the Release live-client target above, then start the evidence runner from the repository root: + +```sh +# First verify the runner and renderer using the local fake server. +node script/native_mac_acceptance.mjs --fake --frames 180 + +# Use the real server's shared host, auth port, and game channel port. +node script/native_mac_acceptance.mjs --server HOST:AUTH_PORT:GAME_PORT +``` + +Real-server mode checks both ports before starting, opens the native login screen, +and collects a redacted client log, one-second process RSS samples, and a JSON +report under `build/native-acceptance/`. Enter credentials in the app, then +exercise login, character selection, movement, combat, map changes, inventory, +chat, window resize/focus, and visual comparison with the current client. Close +the window after at least 30 minutes. The report records whether that minimum +was met, but remains `NEEDS_MANUAL_REVIEW` until those actions and visual results +are checked by a person. It does not assert that RSS alone proves no GPU leak. + +`--live-server` currently accepts one shared host for the auth and game ports. +If those endpoints use different hosts, update the native connection setup +before claiming a real-server pass. Credentials are entered in the client UI; +the runner never puts them in arguments or the report. + +## Run the real 40250 client benchmark in native Vulkan (`--live-client`) + +When built with `port_platform`, `mt_native_render` boots `system.py`, logs in via loopback `FakeLoginServer`, enters the outdoor map with 1..64 monsters, renders the complete 3D scene (40250 hardware-transform terrain splats, animated water patches, gradient skybox & scrolling clouds, SpeedTree forest bark/leaf geometry, GPU-skinned characters with stage-1 specular sphere-maps, and 2D UI/minimap/text-tails/software cursor), and optionally writes a Version 5 `.mtdr` capture: + +```sh +./build-release/native_render/mt_native_render \ + --live-client "/path/to/40250/Server Client TMP4/Client" \ + --fake-mobs 64 --frames 180 --gpu-skinning --no-vsync \ + --capture-out /tmp/mt_full_64mobs.mtdr +``` + +Compare against CPU skinning (`GrannyDeformVertices` on CPU + per-frame vertex buffer re-uploads) with `--no-gpu-skinning`: + +```sh +./build-release/native_render/mt_native_render \ + --live-client "/path/to/40250/Server Client TMP4/Client" \ + --fake-mobs 64 --frames 180 --no-gpu-skinning --no-vsync +``` + +## Replay a captured frame (`.mtdr` v1 / v2 / v3 / v4 / v5) + +```sh +./build-release/native_render/mt_native_render \ + --capture /tmp/mt_full_64mobs.mtdr --frames 180 --animate-bones --no-vsync +``` + +- Pass `--animate-bones` to animate the bone palette SSBO each frame without re-uploading any vertex buffers (`uploads` stays equal to unique static geometries uploaded on frame 0). +- Pass `--animate-first-draw` to increment the first draw's `geometry_revision` each frame after frame 0 and verify incremental GPU buffer re-uploads. + +Capture format Version 5 (backward-compatible with Versions 1–4) stores: +1. 3D draws (`Render3DDraw`): `geometry_key`, `geometry_revision`, matrices, positions, normals, UVs, diffuse colors, indices, D3D8 fixed-function states, `texture0` / `texture1` names, and GPU skinning data (`bone_indices`, `bone_weights`, `bone_matrices`). +2. Self-contained textures: `.dds`, `.tga`, `.jpg`, `.png`, and `.bmp` pack bytes plus `"MTRA"` raw RGBA memory textures (`mem:@` font glyph pages). +3. 2D UI stream (`UIRenderCommand`): canvas size (`ui_width`, `ui_height`) and all `Bar`, `GradientBar`, `Line`, and `Image` commands (including `behind_3d`, clip rects, minimap mask UV coordinates, and software mouse cursor quads). +4. Per-draw fog color, vertex/table mode, range flag, start/end distance and density. Older captures omit these fields and replay without fog. + +The native shader now evaluates recorded D3D8 stage 0/1 color and alpha operations, including the original cloud operation (`D3DTOP_MODULATEINVALPHA_ADDCOLOR = 20`), and applies the captured linear or exponential fog. Expanded UI image modes use the 40250 blend factors for screen/color-dodge and modulate. These state fixes do not by themselves establish pixel parity with a Windows 40250 screenshot; compare the same map, time, camera and UI state before treating a color difference as resolved. diff --git a/native_render/draw_capture.h b/native_render/draw_capture.h new file mode 100644 index 00000000..92c26779 --- /dev/null +++ b/native_render/draw_capture.h @@ -0,0 +1,378 @@ +#pragma once + +#include "../extension/src/platform/EterLib/RenderCommands3D.h" +#include "../extension/src/platform/EterLib/UIRenderCommands.h" + +#include +#include +#include +#include +#include +#include +#include + +// Diagnostic, little-endian arm64 format: +// - Version 1: matrices, positions, diffuse, indices. +// - Version 2: adds stable geometry_key and geometry_revision. +// - Version 3: adds normals, UVs, D3D8 fixed-function states, and embedded pack texture bytes. +// - Version 4: adds GPU skeletal skinning streams (bone_indices, bone_weights, bone_matrices) +// and 2D UIRenderCommand stream + memory textures ("MTRA"). +// - Version 5: adds per-draw fog color, mode, range and distance/density state. +// - Version 6: adds texture address and filtering state for both stages. +// - Version 7: adds two fixed-function lights and material color sources. +// Capture is opt-in and never runs in the normal game path. +namespace native_draw_capture { + +struct Capture { + std::uint32_t version = 7; + std::vector draws; + std::unordered_map> textures; + std::uint32_t ui_width = 960; + std::uint32_t ui_height = 640; + std::vector ui_commands; +}; + +inline std::vector encode_raw_argb_as_mtra( + std::uint32_t width, + std::uint32_t height, + const std::uint32_t* argb) { + std::vector out; + if (!width || !height || !argb) + return out; + const std::size_t pixel_count = std::size_t(width) * std::size_t(height); + out.resize(12 + pixel_count * 4); + out[0] = 'M'; out[1] = 'T'; out[2] = 'R'; out[3] = 'A'; + std::memcpy(out.data() + 4, &width, 4); + std::memcpy(out.data() + 8, &height, 4); + std::uint8_t* dst = out.data() + 12; + for (std::size_t i = 0; i < pixel_count; ++i) { + const std::uint32_t c = argb[i]; + dst[i * 4 + 0] = static_cast((c >> 16) & 0xffu); + dst[i * 4 + 1] = static_cast((c >> 8) & 0xffu); + dst[i * 4 + 2] = static_cast(c & 0xffu); + dst[i * 4 + 3] = static_cast((c >> 24) & 0xffu); + } + return out; +} + +template void write_scalar(std::ofstream& file, const T& value) { + file.write(reinterpret_cast(&value), sizeof(value)); +} + +template void read_scalar(std::ifstream& file, T& value) { + file.read(reinterpret_cast(&value), sizeof(value)); + if (!file) throw std::runtime_error("truncated native draw capture"); +} + +template +void write_vector(std::ofstream& file, const std::vector& values, std::size_t max_count = 4'000'000) { + if (values.size() > max_count) throw std::runtime_error("native draw capture array too large"); + const auto count = static_cast(values.size()); + write_scalar(file, count); + if (count) file.write(reinterpret_cast(values.data()), count * sizeof(T)); +} + +template +void read_vector(std::ifstream& file, std::vector& values, std::size_t max_count = 4'000'000) { + std::uint32_t count = 0; + read_scalar(file, count); + if (count > max_count) throw std::runtime_error("native draw capture array too large"); + values.resize(count); + if (count) file.read(reinterpret_cast(values.data()), count * sizeof(T)); + if (!file) throw std::runtime_error("truncated native draw capture array"); +} + +inline void write_string(std::ofstream& file, const std::string& value) { + if (value.size() > 4096) throw std::runtime_error("native draw capture string too large"); + const auto size = static_cast(value.size()); + write_scalar(file, size); + if (size) file.write(value.data(), size); +} + +inline void read_string(std::ifstream& file, std::string& value) { + std::uint32_t size = 0; + read_scalar(file, size); + if (size > 4096) throw std::runtime_error("native draw capture string too large"); + value.resize(size); + if (size) file.read(value.data(), size); + if (!file) throw std::runtime_error("truncated native draw capture string"); +} + +inline void write( + const std::string& path, + const std::vector& draws, + const std::unordered_map>& textures = {}, + std::uint32_t ui_width = 960, + std::uint32_t ui_height = 640, + const std::vector& ui_commands = {}) { + if (draws.size() > 10'000 || textures.size() > 8'192 || ui_commands.size() > 100'000) + throw std::runtime_error("native draw capture has too many draws, textures, or UI commands"); + std::ofstream file(path, std::ios::binary | std::ios::trunc); + if (!file) throw std::runtime_error("cannot create native draw capture: " + path); + const std::uint32_t magic = 0x4d544452; // MTDR + const std::uint32_t version = 7; + const auto count = static_cast(draws.size()); + write_scalar(file, magic); write_scalar(file, version); write_scalar(file, count); + for (const auto& draw : draws) { + file.write(reinterpret_cast(draw.world), sizeof(draw.world)); + file.write(reinterpret_cast(draw.view), sizeof(draw.view)); + file.write(reinterpret_cast(draw.proj), sizeof(draw.proj)); + write_scalar(file, draw.geometry_key); + write_scalar(file, draw.geometry_revision); + const std::uint32_t flags = + (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u) | (draw.light0 ? 4u : 0u); + write_scalar(file, flags); + write_vector(file, draw.positions); + write_vector(file, draw.diffuse); + write_vector(file, draw.indices); + file.write(reinterpret_cast(draw.viewport), sizeof(draw.viewport)); + write_string(file, draw.texture0); + write_string(file, draw.texture1); + write_vector(file, draw.rhw); + write_vector(file, draw.normals); + write_vector(file, draw.uv0); + write_vector(file, draw.uv1); + write_scalar(file, draw.alpha_blend); + write_scalar(file, draw.src_blend); + write_scalar(file, draw.dest_blend); + write_scalar(file, draw.alpha_test); + write_scalar(file, draw.alpha_ref); + write_scalar(file, draw.alpha_func); + write_scalar(file, draw.cull_mode); + write_scalar(file, draw.z_enable); + write_scalar(file, draw.z_write); + write_scalar(file, draw.z_func); + write_scalar(file, draw.lighting); + write_scalar(file, draw.texture_factor); + write_scalar(file, draw.fog_enable); + file.write(reinterpret_cast(draw.color_op), sizeof(draw.color_op)); + file.write(reinterpret_cast(draw.color_arg1), sizeof(draw.color_arg1)); + file.write(reinterpret_cast(draw.color_arg2), sizeof(draw.color_arg2)); + file.write(reinterpret_cast(draw.alpha_op), sizeof(draw.alpha_op)); + file.write(reinterpret_cast(draw.alpha_arg1), sizeof(draw.alpha_arg1)); + file.write(reinterpret_cast(draw.alpha_arg2), sizeof(draw.alpha_arg2)); + file.write(reinterpret_cast(draw.material_diffuse), sizeof(draw.material_diffuse)); + file.write(reinterpret_cast(draw.material_ambient), sizeof(draw.material_ambient)); + file.write(reinterpret_cast(draw.material_emissive), sizeof(draw.material_emissive)); + file.write(reinterpret_cast(draw.light0_direction), sizeof(draw.light0_direction)); + file.write(reinterpret_cast(draw.light0_diffuse), sizeof(draw.light0_diffuse)); + file.write(reinterpret_cast(draw.light0_ambient), sizeof(draw.light0_ambient)); + write_scalar(file, draw.ambient); + write_scalar(file, draw.fog_color); + write_scalar(file, draw.fog_vertex_mode); + write_scalar(file, draw.fog_table_mode); + write_scalar(file, draw.fog_range_enable); + write_scalar(file, draw.fog_start); + write_scalar(file, draw.fog_end); + write_scalar(file, draw.fog_density); + file.write(reinterpret_cast(draw.address_u), sizeof(draw.address_u)); + file.write(reinterpret_cast(draw.address_v), sizeof(draw.address_v)); + file.write(reinterpret_cast(draw.min_filter), sizeof(draw.min_filter)); + file.write(reinterpret_cast(draw.mag_filter), sizeof(draw.mag_filter)); + file.write(reinterpret_cast(draw.mip_filter), sizeof(draw.mip_filter)); + file.write(reinterpret_cast(draw.lights), sizeof(draw.lights)); + write_scalar(file, draw.diffuse_material_source); + write_scalar(file, draw.ambient_material_source); + write_scalar(file, draw.color_vertex); + write_vector(file, draw.bone_indices); + write_vector(file, draw.bone_weights); + write_vector(file, draw.bone_matrices); + } + const auto texture_count = static_cast(textures.size()); + write_scalar(file, texture_count); + for (const auto& [name, bytes] : textures) { + write_string(file, name); + write_vector(file, bytes, 16'777'216); + } + write_scalar(file, ui_width); + write_scalar(file, ui_height); + const auto ui_count = static_cast(ui_commands.size()); + write_scalar(file, ui_count); + for (const auto& cmd : ui_commands) { + const auto kind = static_cast(cmd.kind); + write_scalar(file, kind); + write_scalar(file, cmd.x1); + write_scalar(file, cmd.y1); + write_scalar(file, cmd.x2); + write_scalar(file, cmd.y2); + write_scalar(file, cmd.argb); + write_scalar(file, cmd.end_argb); + write_scalar(file, cmd.clip_x1); + write_scalar(file, cmd.clip_y1); + write_scalar(file, cmd.clip_x2); + write_scalar(file, cmd.clip_y2); + write_string(file, cmd.text); + const std::uint32_t uiflags = (cmd.quad ? 1u : 0u) | (cmd.behind_3d ? 2u : 0u); + write_scalar(file, uiflags); + file.write(reinterpret_cast(cmd.qx), sizeof(cmd.qx)); + file.write(reinterpret_cast(cmd.qy), sizeof(cmd.qy)); + write_scalar(file, cmd.su); + write_scalar(file, cmd.sv); + write_scalar(file, cmd.eu); + write_scalar(file, cmd.ev); + write_scalar(file, cmd.blend); + write_string(file, cmd.mask); + file.write(reinterpret_cast(cmd.mu), sizeof(cmd.mu)); + file.write(reinterpret_cast(cmd.mv), sizeof(cmd.mv)); + } + if (!file) throw std::runtime_error("failed to write native draw capture: " + path); +} + +inline Capture read_capture(const std::string& path) { + std::ifstream file(path, std::ios::binary); + if (!file) throw std::runtime_error("cannot open native draw capture: " + path); + std::uint32_t magic = 0, version = 0, count = 0; + read_scalar(file, magic); read_scalar(file, version); read_scalar(file, count); + if (magic != 0x4d544452 || (version < 1 || version > 7) || count > 10'000) + throw std::runtime_error("unsupported native draw capture format"); + Capture capture; + capture.version = version; + capture.draws.resize(count); + for (auto& draw : capture.draws) { + file.read(reinterpret_cast(draw.world), sizeof(draw.world)); + file.read(reinterpret_cast(draw.view), sizeof(draw.view)); + file.read(reinterpret_cast(draw.proj), sizeof(draw.proj)); + if (!file) throw std::runtime_error("truncated native draw capture matrices"); + if (version >= 2) { + read_scalar(file, draw.geometry_key); + read_scalar(file, draw.geometry_revision); + } + std::uint32_t flags = 0; + read_scalar(file, flags); + draw.lines = (flags & 1u) != 0; + draw.pretransformed = (flags & 2u) != 0; + draw.light0 = (flags & 4u) != 0; + read_vector(file, draw.positions); + read_vector(file, draw.diffuse); + read_vector(file, draw.indices); + if (version >= 3) { + file.read(reinterpret_cast(draw.viewport), sizeof(draw.viewport)); + read_string(file, draw.texture0); + read_string(file, draw.texture1); + read_vector(file, draw.rhw); + read_vector(file, draw.normals); + read_vector(file, draw.uv0); + read_vector(file, draw.uv1); + read_scalar(file, draw.alpha_blend); + read_scalar(file, draw.src_blend); + read_scalar(file, draw.dest_blend); + read_scalar(file, draw.alpha_test); + read_scalar(file, draw.alpha_ref); + read_scalar(file, draw.alpha_func); + read_scalar(file, draw.cull_mode); + read_scalar(file, draw.z_enable); + read_scalar(file, draw.z_write); + read_scalar(file, draw.z_func); + read_scalar(file, draw.lighting); + read_scalar(file, draw.texture_factor); + read_scalar(file, draw.fog_enable); + file.read(reinterpret_cast(draw.color_op), sizeof(draw.color_op)); + file.read(reinterpret_cast(draw.color_arg1), sizeof(draw.color_arg1)); + file.read(reinterpret_cast(draw.color_arg2), sizeof(draw.color_arg2)); + file.read(reinterpret_cast(draw.alpha_op), sizeof(draw.alpha_op)); + file.read(reinterpret_cast(draw.alpha_arg1), sizeof(draw.alpha_arg1)); + file.read(reinterpret_cast(draw.alpha_arg2), sizeof(draw.alpha_arg2)); + file.read(reinterpret_cast(draw.material_diffuse), sizeof(draw.material_diffuse)); + file.read(reinterpret_cast(draw.material_ambient), sizeof(draw.material_ambient)); + file.read(reinterpret_cast(draw.material_emissive), sizeof(draw.material_emissive)); + file.read(reinterpret_cast(draw.light0_direction), sizeof(draw.light0_direction)); + file.read(reinterpret_cast(draw.light0_diffuse), sizeof(draw.light0_diffuse)); + file.read(reinterpret_cast(draw.light0_ambient), sizeof(draw.light0_ambient)); + read_scalar(file, draw.ambient); + if (version >= 5) { + read_scalar(file, draw.fog_color); + read_scalar(file, draw.fog_vertex_mode); + read_scalar(file, draw.fog_table_mode); + read_scalar(file, draw.fog_range_enable); + read_scalar(file, draw.fog_start); + read_scalar(file, draw.fog_end); + read_scalar(file, draw.fog_density); + } + if (version >= 6) { + file.read(reinterpret_cast(draw.address_u), sizeof(draw.address_u)); + file.read(reinterpret_cast(draw.address_v), sizeof(draw.address_v)); + file.read(reinterpret_cast(draw.min_filter), sizeof(draw.min_filter)); + file.read(reinterpret_cast(draw.mag_filter), sizeof(draw.mag_filter)); + file.read(reinterpret_cast(draw.mip_filter), sizeof(draw.mip_filter)); + } + if (version >= 7) { + file.read(reinterpret_cast(draw.lights), sizeof(draw.lights)); + read_scalar(file, draw.diffuse_material_source); + read_scalar(file, draw.ambient_material_source); + read_scalar(file, draw.color_vertex); + } else if (draw.light0) { + auto& light = draw.lights[0]; + light.type = 3; + std::memcpy(light.direction, draw.light0_direction, sizeof(light.direction)); + std::memcpy(light.diffuse, draw.light0_diffuse, sizeof(light.diffuse)); + std::memcpy(light.ambient, draw.light0_ambient, sizeof(light.ambient)); + } + if (!file) throw std::runtime_error("truncated native draw capture state"); + } else { + draw.z_enable = 1; + draw.z_write = 1; + } + if (version >= 4) { + read_vector(file, draw.bone_indices); + read_vector(file, draw.bone_weights); + read_vector(file, draw.bone_matrices); + } + } + if (version >= 3) { + std::uint32_t texture_count = 0; + read_scalar(file, texture_count); + if (texture_count > 8'192) throw std::runtime_error("native draw capture has too many textures"); + for (std::uint32_t i = 0; i < texture_count; ++i) { + std::string name; + std::vector bytes; + read_string(file, name); + read_vector(file, bytes, 16'777'216); + capture.textures.emplace(std::move(name), std::move(bytes)); + } + } + if (version >= 4) { + read_scalar(file, capture.ui_width); + read_scalar(file, capture.ui_height); + std::uint32_t ui_count = 0; + read_scalar(file, ui_count); + if (ui_count > 100'000) throw std::runtime_error("native draw capture has too many UI commands"); + capture.ui_commands.resize(ui_count); + for (auto& cmd : capture.ui_commands) { + std::uint32_t kind = 0, uiflags = 0; + read_scalar(file, kind); + cmd.kind = static_cast(kind); + read_scalar(file, cmd.x1); + read_scalar(file, cmd.y1); + read_scalar(file, cmd.x2); + read_scalar(file, cmd.y2); + read_scalar(file, cmd.argb); + read_scalar(file, cmd.end_argb); + read_scalar(file, cmd.clip_x1); + read_scalar(file, cmd.clip_y1); + read_scalar(file, cmd.clip_x2); + read_scalar(file, cmd.clip_y2); + read_string(file, cmd.text); + read_scalar(file, uiflags); + cmd.quad = (uiflags & 1u) != 0; + cmd.behind_3d = (uiflags & 2u) != 0; + file.read(reinterpret_cast(cmd.qx), sizeof(cmd.qx)); + file.read(reinterpret_cast(cmd.qy), sizeof(cmd.qy)); + read_scalar(file, cmd.su); + read_scalar(file, cmd.sv); + read_scalar(file, cmd.eu); + read_scalar(file, cmd.ev); + read_scalar(file, cmd.blend); + read_string(file, cmd.mask); + file.read(reinterpret_cast(cmd.mu), sizeof(cmd.mu)); + file.read(reinterpret_cast(cmd.mv), sizeof(cmd.mv)); + if (!file) throw std::runtime_error("truncated native draw capture UI command"); + } + } + return capture; +} + +inline std::vector read(const std::string& path) { + return read_capture(path).draws; +} + +} // namespace native_draw_capture diff --git a/native_render/main.cpp b/native_render/main.cpp new file mode 100644 index 00000000..f772c885 --- /dev/null +++ b/native_render/main.cpp @@ -0,0 +1,3376 @@ +#include "RenderCommands3D.h" +#include "UIRenderCommands.h" +#include "draw_capture.h" +#include "dxt.h" +#include "touch_controller.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "platform/MilesLib/AudioCommands.h" +#include "platform/PackBackend.h" +#include "platform/ScriptLib/PythonBoot.h" +#include "platform/UserInterface/ServerClock.h" +#include "../tests/port_login_flow_server.h" +#include +#endif + +#include + +#include +#include +#include +#include +#include "stb_image.h" + +#ifdef __APPLE__ +#include +#endif + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { + +void check(VkResult result, const char* operation) { + if (result != VK_SUCCESS) throw std::runtime_error(std::string(operation) + ": VkResult " + std::to_string(result)); +} + +struct Vertex { + float position[3]; + float normal[3]; + float uv[2]; + float color[4]; + std::uint8_t joints[4]; + float weights[4]; + float mask_uv[2]; + float rhw = 1.0f; +}; + +struct PushConstants { + std::array mvp{}; + std::array tint_color{1.0f, 1.0f, 1.0f, 1.0f}; + std::array ambient_emissive{1.0f, 1.0f, 1.0f, -1.0f}; + std::array light_dir{0.0f, 0.0f, 1.0f, 0.0f}; + std::array light_diffuse{0.0f, 0.0f, 0.0f, 0.0f}; +}; +static_assert(sizeof(PushConstants) == 128, "PushConstants must fit within Vulkan's 128-byte minimum guarantee"); + +// std430 payload selected with a dynamic storage-buffer offset for every draw. +// The existing 128-byte push constants cannot hold the D3D8 texture stages and fog state. +struct alignas(16) FixedFunctionState { + std::array stage0_color{}; // op, arg1, arg2, unused + std::array stage0_alpha{}; // op, arg1, arg2, unused + std::array stage1_color{}; + std::array stage1_alpha{}; + std::array fog_color{}; + std::array fog_params{}; // start, end, density, range enabled + std::array texture_factor{}; + std::array world_view{}; + std::array flags{}; // fixed function, texture0, texture1, fog mode + std::array world{}; + std::array lighting_flags{}; // enabled, COLORVERTEX, ambient source, diffuse source + std::array material_ambient{}; + std::array material_emissive{}; + std::array global_ambient{}; + std::array, 2> light_position_type{}; + std::array, 2> light_direction_range{}; + std::array, 2> light_diffuse{}; + std::array, 2> light_ambient{}; + std::array, 2> light_attenuation{}; + std::array, 2> light_spot{}; +}; +static_assert(sizeof(FixedFunctionState) == 512); + +constexpr std::array kIdentityMatrix = { + 1.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 1.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 1.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 1.0f}; + +std::array multiply(const float* a, const float* b) { + std::array result{}; + for (int row = 0; row < 4; ++row) + for (int col = 0; col < 4; ++col) + for (int k = 0; k < 4; ++k) + result[row * 4 + col] += a[row * 4 + k] * b[k * 4 + col]; + return result; +} + +std::array draw_mvp(const Render3DDraw& draw) { + const auto world_view = multiply(draw.world, draw.view); + return multiply(world_view.data(), draw.proj); +} + +bool draw_is_specular_spheremap(const Render3DDraw& draw) { + // D3DTOP_MODULATEALPHA_ADDCOLOR == 18 on texture stage 1 + return !draw.texture1.empty() && draw.color_op[1] == 18; +} + +std::array unpack_argb(std::uint32_t argb) { + return { + float((argb >> 16) & 255) / 255.0f, + float((argb >> 8) & 255) / 255.0f, + float(argb & 255) / 255.0f, + float((argb >> 24) & 255) / 255.0f}; +} + +PushConstants make_push_constants(const Render3DDraw& draw, float skin_offset_encoded) { + PushConstants constants{}; + constants.mvp = draw_mvp(draw); + const bool lit = draw.lighting != 0; + for (int c = 0; c < 4; ++c) { + const float base = lit ? draw.material_diffuse[c] : 1.0f; + constants.tint_color[c] = base; + } + const float alpha_cutoff = draw.alpha_test ? (float(draw.alpha_ref) / 255.0f) : -1.0f; + if (lit && draw.light0) { + const auto amb = unpack_argb(draw.ambient); + for (int c = 0; c < 3; ++c) + constants.ambient_emissive[c] = + draw.material_ambient[c] * std::min(amb[c] + draw.light0_ambient[c], 1.0f) + draw.material_emissive[c]; + constants.ambient_emissive[3] = alpha_cutoff; + const float lx = -draw.light0_direction[0]; + const float ly = -draw.light0_direction[1]; + const float lz = -draw.light0_direction[2]; + float mx = draw.world[0] * lx + draw.world[1] * ly + draw.world[2] * lz; + float my = draw.world[4] * lx + draw.world[5] * ly + draw.world[6] * lz; + float mz = draw.world[8] * lx + draw.world[9] * ly + draw.world[10] * lz; + const float len = std::sqrt(mx * mx + my * my + mz * mz); + if (len > 1e-6f) { + mx /= len; my /= len; mz /= len; + } else { + mx = 0.0f; my = 0.0f; mz = 1.0f; + } + constants.light_dir = {mx, my, mz, 1.0f}; + constants.light_diffuse = { + draw.light0_diffuse[0], draw.light0_diffuse[1], draw.light0_diffuse[2], skin_offset_encoded}; + } else if (lit) { + const auto amb = unpack_argb(draw.ambient); + for (int c = 0; c < 3; ++c) + constants.ambient_emissive[c] = + draw.material_ambient[c] * amb[c] + draw.material_emissive[c]; + constants.ambient_emissive[3] = alpha_cutoff; + constants.light_dir = {0.0f, 0.0f, 1.0f, 1.0f}; + constants.light_diffuse[3] = skin_offset_encoded; + } else { + constants.ambient_emissive = {1.0f, 1.0f, 1.0f, alpha_cutoff}; + constants.light_diffuse[3] = skin_offset_encoded; + } + if (draw_is_specular_spheremap(draw)) { + constants.light_dir[3] = 2.0f; + } else if (!draw.texture1.empty() && (draw.color_op[1] > 1 || draw.alpha_op[1] > 1)) { + constants.light_dir[3] = lit ? -0.6f : -1.0f; + } + if (draw.pretransformed) + constants.light_diffuse[3] = -1.0f; + return constants; +} + +FixedFunctionState make_fixed_function_state(const Render3DDraw& draw) { + FixedFunctionState state{}; + for (int stage = 0; stage < 2; ++stage) { + auto& color = stage ? state.stage1_color : state.stage0_color; + auto& alpha = stage ? state.stage1_alpha : state.stage0_alpha; + color = {draw.color_op[stage], draw.color_arg1[stage], draw.color_arg2[stage], 0}; + alpha = {draw.alpha_op[stage], draw.alpha_arg1[stage], draw.alpha_arg2[stage], + stage == 0 ? draw.alpha_func : draw.alpha_test}; + } + state.texture_factor = unpack_argb(draw.texture_factor); + state.world_view = multiply(draw.world, draw.view); + const auto fog = unpack_argb(draw.fog_color); + state.fog_color = fog; + state.fog_params = {draw.fog_start, draw.fog_end, draw.fog_density, + draw.fog_range_enable ? 1.0f : 0.0f}; + const std::uint32_t fog_mode = draw.fog_enable + ? (draw.fog_table_mode ? draw.fog_table_mode : draw.fog_vertex_mode) : 0; + const bool multiplicative_shadow = draw.alpha_blend && draw.src_blend == 1 && + draw.dest_blend == 3 && !draw.texture0.empty() && draw.color_op[0] == 4; + state.flags = {multiplicative_shadow ? 2u : 1u, !draw.texture0.empty() ? 1u : 0u, + !draw.texture1.empty() ? 1u : 0u, fog_mode}; + std::copy(draw.world, draw.world + 16, state.world.begin()); + state.lighting_flags = {draw.lighting, draw.color_vertex, + draw.ambient_material_source, draw.diffuse_material_source}; + std::copy(draw.material_ambient, draw.material_ambient + 4, state.material_ambient.begin()); + std::copy(draw.material_emissive, draw.material_emissive + 4, state.material_emissive.begin()); + state.global_ambient = unpack_argb(draw.ambient); + for (int i = 0; i < 2; ++i) { + const auto& light = draw.lights[i]; + state.light_position_type[i] = {light.position[0], light.position[1], light.position[2], float(light.type)}; + state.light_direction_range[i] = {light.direction[0], light.direction[1], light.direction[2], light.range}; + std::copy(light.diffuse, light.diffuse + 4, state.light_diffuse[i].begin()); + std::copy(light.ambient, light.ambient + 4, state.light_ambient[i].begin()); + state.light_attenuation[i] = {light.attenuation[0], light.attenuation[1], light.attenuation[2], light.falloff}; + state.light_spot[i] = {light.theta, light.phi, 0, 0}; + } + return state; +} + +std::uint64_t hash_bytes(std::uint64_t hash, const void* bytes, std::size_t size) { + const auto* data = static_cast(bytes); + std::size_t i = 0; + for (; i + sizeof(std::uint64_t) <= size; i += sizeof(std::uint64_t)) { + std::uint64_t word = 0; + std::memcpy(&word, data + i, sizeof(word)); + hash = (hash ^ word) * 1099511628211ull; + } + for (; i < size; ++i) hash = (hash ^ data[i]) * 1099511628211ull; + hash = (hash ^ size) * 1099511628211ull; + return hash; +} + +// Immediate-mode draws have no source-buffer key. For them, verify content before reusing a slot. +std::uint64_t geometry_hash(const Render3DDraw& draw) { + std::uint64_t hash = 14695981039346656037ull; + const std::uint32_t flags = (draw.lines ? 1u : 0u) | (draw.pretransformed ? 2u : 0u); + hash = hash_bytes(hash, &flags, sizeof(flags)); + hash = hash_bytes(hash, draw.positions.data(), draw.positions.size() * sizeof(float)); + hash = hash_bytes(hash, draw.rhw.data(), draw.rhw.size() * sizeof(float)); + hash = hash_bytes(hash, draw.normals.data(), draw.normals.size() * sizeof(float)); + hash = hash_bytes(hash, draw.uv0.data(), draw.uv0.size() * sizeof(float)); + hash = hash_bytes(hash, draw.uv1.data(), draw.uv1.size() * sizeof(float)); + hash = hash_bytes(hash, draw.diffuse.data(), draw.diffuse.size() * sizeof(std::uint32_t)); + return hash_bytes(hash, draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); +} + +mtgodot::Image decode_tga(const std::uint8_t* data, std::size_t size) { + mtgodot::Image out; + if (size < 18) return out; + const std::uint8_t id_len = data[0]; + const std::uint8_t cmap_type = data[1]; + const std::uint8_t img_type = data[2]; + const std::uint32_t w = std::uint32_t(data[12]) | (std::uint32_t(data[13]) << 8); + const std::uint32_t h = std::uint32_t(data[14]) | (std::uint32_t(data[15]) << 8); + const std::uint8_t bpp = data[16]; + const std::uint8_t desc = data[17]; + if (cmap_type != 0 || !w || !h || w > 4096 || h > 4096) return out; + if (img_type != 2 && img_type != 3 && img_type != 10) return out; + const std::size_t bytes_per_pixel = bpp / 8; + if (bytes_per_pixel != 1 && bytes_per_pixel != 3 && bytes_per_pixel != 4) return out; + std::size_t offset = 18 + std::size_t(id_len); + if (offset > size) return out; + + const std::size_t pixel_count = std::size_t(w) * std::size_t(h); + std::vector temp(pixel_count * 4); + auto write_pixel = [&](std::size_t idx, const std::uint8_t* src) { + std::uint8_t* dst = &temp[idx * 4]; + if (bytes_per_pixel == 1) { + dst[0] = dst[1] = dst[2] = src[0]; + dst[3] = 255; + } else if (bytes_per_pixel == 3) { + dst[0] = src[2]; + dst[1] = src[1]; + dst[2] = src[0]; + dst[3] = 255; + } else { + dst[0] = src[2]; + dst[1] = src[1]; + dst[2] = src[0]; + dst[3] = src[3]; + } + }; + + if (img_type == 2 || img_type == 3) { + if (offset + pixel_count * bytes_per_pixel > size) return out; + for (std::size_t i = 0; i < pixel_count; ++i) + write_pixel(i, data + offset + i * bytes_per_pixel); + } else if (img_type == 10) { + std::size_t i = 0; + while (i < pixel_count && offset < size) { + const std::uint8_t header = data[offset++]; + const std::size_t run = (header & 0x7fu) + 1u; + if (header & 0x80u) { + if (offset + bytes_per_pixel > size) return out; + for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) + write_pixel(i, data + offset); + offset += bytes_per_pixel; + } else { + if (offset + run * bytes_per_pixel > size) return out; + for (std::size_t r = 0; r < run && i < pixel_count; ++r, ++i) { + write_pixel(i, data + offset); + offset += bytes_per_pixel; + } + } + } + if (i != pixel_count) return out; + } + + out.w = static_cast(w); + out.h = static_cast(h); + const bool top_origin = (desc & 0x20u) != 0; + if (top_origin) { + out.rgba = std::move(temp); + } else { + out.rgba.resize(pixel_count * 4); + const std::size_t row_bytes = std::size_t(w) * 4; + for (std::uint32_t y = 0; y < h; ++y) + std::memcpy(&out.rgba[std::size_t(y) * row_bytes], &temp[std::size_t(h - 1 - y) * row_bytes], row_bytes); + } + return out; +} + +mtgodot::Image decode_texture_bytes(const std::uint8_t* data, std::size_t size) { + if (!data || size < 12) return {}; + if (data[0] == 'M' && data[1] == 'T' && data[2] == 'R' && data[3] == 'A') { + std::uint32_t w = 0, h = 0; + std::memcpy(&w, data + 4, 4); + std::memcpy(&h, data + 8, 4); + const std::size_t bytes = std::size_t(w) * std::size_t(h) * 4; + if (w > 0 && h > 0 && w <= 4096 && h <= 4096 && size == 12 + bytes) { + mtgodot::Image out; + out.w = static_cast(w); + out.h = static_cast(h); + out.rgba.assign(data + 12, data + 12 + bytes); + return out; + } + return {}; + } + if (data[0] == 'D' && data[1] == 'D' && data[2] == 'S' && data[3] == ' ') + return mtgodot::load_dds(data, size); + auto tga = decode_tga(data, size); + if (tga.ok()) return tga; + if (size > static_cast(INT_MAX)) return {}; + int w = 0, h = 0, channels = 0; + stbi_uc* pixels = stbi_load_from_memory(data, static_cast(size), &w, &h, &channels, 4); + if (pixels && w > 0 && h > 0 && w <= 4096 && h <= 4096) { + mtgodot::Image out; + out.w = static_cast(w); + out.h = static_cast(h); + out.rgba.assign(pixels, pixels + std::size_t(w) * std::size_t(h) * 4); + stbi_image_free(pixels); + return out; + } + stbi_image_free(pixels); + return {}; +} + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +bool read_live_pack_texture(const std::string& vpath, std::vector& bytes) { + if (vpath.empty() || !mtpack40250::ready()) + return false; + std::string norm = vpath; + for (char& ch : norm) + if (ch == '\\') ch = '/'; + std::string stripped = norm; + if (stripped.size() >= 2 && stripped[1] == ':') + stripped = stripped.substr(2); + while (!stripped.empty() && stripped.front() == '/') + stripped.erase(stripped.begin()); + auto lower = [](std::string s) { + for (char& ch : s) + if (ch >= 'A' && ch <= 'Z') + ch = static_cast(ch - 'A' + 'a'); + return s; + }; + for (const std::string& candidate : { + norm, stripped, "d:/" + stripped, + lower(norm), lower(stripped), lower("d:/" + stripped)}) { + if (mtpack40250::read(candidate, bytes) && !bytes.empty()) + return true; + } + return false; +} + +SDL_Cursor* load_game_cursor(int shape) { + static constexpr std::array images = { + "cursor.sub", "cursor_attack.sub", "cursor_attack.sub", "cursor_talk.sub", + "cursor_no.sub", "cursor_pick.sub", "cursor_door.sub", "cursor_chair.sub", + "cursor_chair.sub", "cursor_buy.sub", "cursor_sell.sub", + "cursor_camera_rotate.sub", "cursor_hsize.sub", "cursor_vsize.sub", "cursor_hvsize.sub"}; + if (shape < 0 || shape >= static_cast(images.size())) return nullptr; + const std::string sub_path = std::string("d:/ymir work/ui/cursor/") + images[shape]; + std::vector sub_bytes; + if (!read_live_pack_texture(sub_path, sub_bytes)) return nullptr; + std::unordered_map tokens; + std::istringstream input(std::string(sub_bytes.begin(), sub_bytes.end())); + std::string line; + while (std::getline(input, line)) { + std::istringstream fields(line); + std::string key, value; + if (!(fields >> key >> value)) continue; + if (!value.empty() && value.front() == '"') { + value.erase(0, 1); + if (!value.empty() && value.back() == '"') value.pop_back(); + } + std::transform(key.begin(), key.end(), key.begin(), [](unsigned char ch) { return std::tolower(ch); }); + tokens[key] = value; + } + if (tokens["title"] != "subimage" || tokens["image"].empty()) return nullptr; + const std::string image_path = tokens["version"] == "2.0" + ? std::string("d:/ymir work/ui/cursor/") + tokens["image"] + : std::string("d:/ymir work/ui/") + tokens["image"]; + std::vector image_bytes; + if (!read_live_pack_texture(image_path, image_bytes)) return nullptr; + const auto image = decode_texture_bytes(image_bytes.data(), image_bytes.size()); + if (!image.ok()) return nullptr; + const int left = std::atoi(tokens["left"].c_str()); + const int top = std::atoi(tokens["top"].c_str()); + const int right = std::atoi(tokens["right"].c_str()); + const int bottom = std::atoi(tokens["bottom"].c_str()); + if (left < 0 || top < 0 || right <= left || bottom <= top || + right > image.w || bottom > image.h) return nullptr; + const int width = right - left, height = bottom - top; + std::vector pixels(std::size_t(width) * height * 4); + for (int y = 0; y < height; ++y) + std::memcpy(pixels.data() + std::size_t(y) * width * 4, + image.rgba.data() + (std::size_t(top + y) * image.w + left) * 4, + std::size_t(width) * 4); + SDL_Surface* surface = SDL_CreateSurfaceFrom(width, height, SDL_PIXELFORMAT_RGBA32, + pixels.data(), width * 4); + if (!surface) return nullptr; + const int hot_x = shape >= 12 ? std::min(16, width - 1) : 0; + const int hot_y = shape >= 12 ? std::min(16, height - 1) : 0; + SDL_Cursor* cursor = SDL_CreateColorCursor(surface, hot_x, hot_y); + SDL_DestroySurface(surface); + return cursor; +} + +class NativeAudioEngine { + struct MemoryAudioBuffer { + const std::uint8_t* data = nullptr; + std::size_t size = 0; + }; + struct Voice { + const std::vector* samples = nullptr; + std::size_t frame_cursor = 0; + float volume = 1.0f; + float target_volume = 1.0f; + float fade_step_per_frame = 0.0f; + bool stop_after_fade = false; + bool loop = false; + bool is_3d = false; + int handle = 0; + std::string filename; + }; + +#ifdef __APPLE__ + static OSStatus mem_audio_read_proc( + void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) { + const auto* mem = static_cast(inClientData); + if (inPosition < 0 || static_cast(inPosition) >= mem->size) { + *actualCount = 0; + return noErr; + } + const std::size_t avail = mem->size - static_cast(inPosition); + const std::size_t to_read = std::min(requestCount, avail); + std::memcpy(buffer, mem->data + inPosition, to_read); + *actualCount = static_cast(to_read); + return noErr; + } + + static SInt64 mem_audio_get_size_proc(void* inClientData) { + return static_cast(static_cast(inClientData)->size); + } +#endif + +public: + NativeAudioEngine() { + if (SDL_InitSubSystem(SDL_INIT_AUDIO)) { + SDL_AudioSpec spec{}; + spec.format = SDL_AUDIO_F32; + spec.channels = 2; + spec.freq = 44100; + stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_PLAYBACK, &spec, nullptr, nullptr); + if (stream_) + SDL_ResumeAudioStreamDevice(stream_); + } + // Always DrainAudioCommands() from cold boot so stale queued commands don't accumulate. + (void)DrainAudioCommands(); + } + + ~NativeAudioEngine() { + if (stream_) + SDL_DestroyAudioStream(stream_); + SDL_QuitSubSystem(SDL_INIT_AUDIO); + } + + void pump() { + const auto commands = DrainAudioCommands(); + for (const auto& cmd : commands) { + ++commands_drained_; + switch (cmd.type) { + case AudioCommand::PlaySound2D: { + const auto* clip = get_clip(cmd.filename); + if (clip && !clip->empty()) { + Voice v{}; + v.samples = clip; + v.volume = sound_volume_; + v.target_volume = sound_volume_; + v.filename = cmd.filename; + voices_.push_back(std::move(v)); + ++played_2d_; + } + break; + } + case AudioCommand::PlaySound3D: { + const auto* clip = get_clip(cmd.filename); + if (clip && !clip->empty()) { + Voice v{}; + v.samples = clip; + v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + v.target_volume = v.volume; + v.loop = cmd.play_count != 1; + v.is_3d = true; + v.handle = cmd.id; + v.filename = cmd.filename; + voices_.push_back(std::move(v)); + ++played_3d_; + } + break; + } + case AudioCommand::StopSound3D: + voices_.erase( + std::remove_if(voices_.begin(), voices_.end(), [&](const Voice& v) { + return v.is_3d && v.handle == cmd.id; + }), + voices_.end()); + break; + case AudioCommand::StopAllSound3D: + voices_.erase( + std::remove_if(voices_.begin(), voices_.end(), [](const Voice& v) { return v.is_3d; }), + voices_.end()); + break; + case AudioCommand::SetSoundVolume3D: + for (auto& v : voices_) { + if (v.is_3d && v.handle == cmd.id) { + v.volume = sound_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + v.target_volume = v.volume; + } + } + break; + case AudioCommand::PlayMusic: + case AudioCommand::FadeInMusic: { + const auto* clip = get_clip(cmd.filename); + if (clip && !clip->empty()) { + bgm_.samples = clip; + bgm_.frame_cursor = 0; + bgm_.loop = true; + bgm_.filename = cmd.filename; + bgm_.stop_after_fade = false; + const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + bgm_.target_volume = target; + if (cmd.type == AudioCommand::FadeInMusic) { + bgm_.volume = 0.0f; + bgm_.fade_step_per_frame = target / (44100.0f * 1.5f); + } else { + bgm_.volume = target; + bgm_.fade_step_per_frame = 0.0f; + } + ++played_music_; + } + break; + } + case AudioCommand::FadeOutMusic: + case AudioCommand::FadeOutAllMusic: + if (bgm_.samples) { + bgm_.target_volume = 0.0f; + bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f); + bgm_.stop_after_fade = true; + } + break; + case AudioCommand::FadeLimitOutMusic: + if (bgm_.samples) { + bgm_.target_volume = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); + bgm_.fade_step_per_frame = (bgm_.target_volume - bgm_.volume) / (44100.0f * 1.0f); + bgm_.stop_after_fade = false; + } + break; + case AudioCommand::SetMusicVolume: + music_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); + if (bgm_.samples && !bgm_.stop_after_fade) { + bgm_.volume = music_volume_; + bgm_.target_volume = music_volume_; + } + break; + case AudioCommand::SetSoundVolume: + sound_volume_ = std::clamp(cmd.volume, 0.0f, 1.0f); + break; + default: + break; + } + } + + if (!stream_) return; + const int queued_bytes = SDL_GetAudioStreamAvailable(stream_); + if (queued_bytes < 0) return; + const std::size_t queued_frames = static_cast(queued_bytes) / (2 * sizeof(float)); + constexpr std::size_t kTargetQueuedFrames = 4410; // ~100 ms stereo @ 44.1 kHz + if (queued_frames >= kTargetQueuedFrames) return; + const std::size_t frames_to_mix = kTargetQueuedFrames - queued_frames; + mix_buffer_.assign(frames_to_mix * 2, 0.0f); + + auto mix_voice = [&](Voice& v) -> bool { + if (!v.samples || v.samples->empty()) return false; + const std::size_t total_frames = v.samples->size() / 2; + if (total_frames == 0) return false; + const float* src = v.samples->data(); + for (std::size_t f = 0; f < frames_to_mix; ++f) { + if (v.frame_cursor >= total_frames) { + if (v.loop) v.frame_cursor = 0; + else return false; + } + if (v.fade_step_per_frame != 0.0f) { + v.volume += v.fade_step_per_frame; + if ((v.fade_step_per_frame > 0.0f && v.volume >= v.target_volume) || + (v.fade_step_per_frame < 0.0f && v.volume <= v.target_volume)) { + v.volume = v.target_volume; + v.fade_step_per_frame = 0.0f; + if (v.stop_after_fade && v.volume <= 0.0001f) { + v.samples = nullptr; + return false; + } + } + } + mix_buffer_[f * 2 + 0] += src[v.frame_cursor * 2 + 0] * v.volume; + mix_buffer_[f * 2 + 1] += src[v.frame_cursor * 2 + 1] * v.volume; + ++v.frame_cursor; + } + return v.loop || v.frame_cursor < total_frames; + }; + + if (bgm_.samples) { + if (!mix_voice(bgm_)) bgm_.samples = nullptr; + } + for (auto it = voices_.begin(); it != voices_.end();) { + if (!mix_voice(*it)) it = voices_.erase(it); + else ++it; + } + for (float& sample : mix_buffer_) + sample = std::clamp(sample, -1.0f, 1.0f); + SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast(mix_buffer_.size() * sizeof(float))); + } + +private: + const std::vector* get_clip(const std::string& vpath) { + if (vpath.empty()) return nullptr; + auto [it, inserted] = clips_.try_emplace(vpath); + if (!inserted) return &it->second; +#ifdef __APPLE__ + std::vector bytes; + if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16) + return &it->second; + MemoryAudioBuffer mem{bytes.data(), bytes.size()}; + AudioFileID audio_file = nullptr; + if (AudioFileOpenWithCallbacks( + &mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr || + !audio_file) { + return &it->second; + } + ExtAudioFileRef ext_file = nullptr; + if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) { + AudioFileClose(audio_file); + return &it->second; + } + AudioStreamBasicDescription client_format{}; + client_format.mSampleRate = 44100.0; + client_format.mFormatID = kAudioFormatLinearPCM; + client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked; + client_format.mBytesPerPacket = 8; + client_format.mFramesPerPacket = 1; + client_format.mBytesPerFrame = 8; + client_format.mChannelsPerFrame = 2; + client_format.mBitsPerChannel = 32; + if (ExtAudioFileSetProperty( + ext_file, + kExtAudioFileProperty_ClientDataFormat, + sizeof(client_format), + &client_format) == noErr) { + std::vector chunk(4096 * 2); + while (true) { + UInt32 frame_count = 4096; + AudioBufferList buf_list{}; + buf_list.mNumberBuffers = 1; + buf_list.mBuffers[0].mNumberChannels = 2; + buf_list.mBuffers[0].mDataByteSize = static_cast(chunk.size() * sizeof(float)); + buf_list.mBuffers[0].mData = chunk.data(); + if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0) + break; + it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2); + if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes + break; + } + } + ExtAudioFileDispose(ext_file); + AudioFileClose(audio_file); +#endif + return &it->second; + } + + SDL_AudioStream* stream_ = nullptr; + std::unordered_map> clips_; + std::vector voices_; + Voice bgm_{}; + std::vector mix_buffer_; + float sound_volume_ = 0.8f; + float music_volume_ = 0.5f; + std::size_t commands_drained_ = 0; + std::size_t played_2d_ = 0; + std::size_t played_3d_ = 0; + std::size_t played_music_ = 0; +}; +#endif + +std::string bundled_file_path(const char* name) { + const char* base = SDL_GetBasePath(); + return std::string(base ? base : "./") + name; +} + +std::vector read_spirv(const char* name) { + std::size_t size = 0; + void* bytes = SDL_LoadFile(bundled_file_path(name).c_str(), &size); +#ifdef __ANDROID__ + if (!bytes) bytes = SDL_LoadFile((std::string("assets://") + name).c_str(), &size); + if (!bytes) bytes = SDL_LoadFile(name, &size); +#endif + if (!bytes) throw std::runtime_error(std::string("cannot open bundled shader: ") + name); + if (size == 0 || size % 4) { + SDL_free(bytes); + throw std::runtime_error(std::string("invalid SPIR-V size: ") + name); + } + std::vector words(size / 4); + std::memcpy(words.data(), bytes, size); + SDL_free(bytes); + return words; +} + +class VulkanWindow { + struct GeometryId { + std::uint64_t key = 0, signature = 0; + bool operator==(const GeometryId&) const = default; + }; + struct GeometryIdHash { + std::size_t operator()(GeometryId id) const { + return std::size_t(id.key ^ (id.signature + 0x9e3779b97f4a7c15ull + (id.key << 6) + (id.key >> 2))); + } + }; + struct Geometry { + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + VkDeviceSize index_offset = 0; + std::uint64_t last_used_frame = 0; + std::uint32_t vertex_count = 0, index_count = 0; + }; + struct GpuTexture { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + VkImageView view = VK_NULL_HANDLE; + VkDescriptorSet descriptor = VK_NULL_HANDLE; + VkDescriptorSet ui_descriptor = VK_NULL_HANDLE; + std::uint64_t last_used_frame = 0; + std::uint64_t last_decode_attempt_frame = 0; + }; + struct PairedDescriptor { + VkDescriptorSet set = VK_NULL_HANDLE; + std::uint64_t last_used_frame = 0; + }; + struct PreparedDraw { + Geometry* geometry = nullptr; + VkPipeline pipeline = VK_NULL_HANDLE; + VkDescriptorSet descriptor = VK_NULL_HANDLE; + PushConstants constants{}; + FixedFunctionState fixed_state{}; + std::uint32_t state_offset = 0; + }; + struct UiBatch { + std::uint32_t first_index = 0; + std::uint32_t index_count = 0; + VkPipeline pipeline = VK_NULL_HANDLE; + VkDescriptorSet descriptor = VK_NULL_HANDLE; + PushConstants constants{}; + bool behind_3d = false; + std::uint32_t state_offset = 0; + }; + + static constexpr std::size_t kMaxBonesPerFrame = 65536; + static constexpr VkDeviceSize kBoneBufferBytes = kMaxBonesPerFrame * 16 * sizeof(float); + static constexpr std::size_t kMaxUiVerticesPerFrame = 65536; + static constexpr std::size_t kMaxUiIndicesPerFrame = 98304; + static constexpr VkDeviceSize kUiVertexBytes = kMaxUiVerticesPerFrame * sizeof(Vertex); + static constexpr VkDeviceSize kUiBufferBytes = kUiVertexBytes + kMaxUiIndicesPerFrame * sizeof(std::uint32_t); + +public: + struct Timings { + double sync_ms = 0; + double prepare_ms = 0; + double steady_prepare_ms = 0; + double submit_ms = 0; + double present_ms = 0; + double gpu_ms = 0; + std::uint64_t gpu_samples = 0; + }; + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + void enable_game_hardware_cursor() { + hardware_cursor_enabled_ = true; + for (int shape = 0; shape < static_cast(game_cursors_.size()); ++shape) { + game_cursors_[shape] = load_game_cursor(shape); + } + fallback_cursor_ = SDL_CreateSystemCursor(SDL_SYSTEM_CURSOR_DEFAULT); + SDL_SetCursor(game_cursors_[0] ? game_cursors_[0] : fallback_cursor_); + SDL_ShowCursor(); + os_cursor_hidden_ = false; + } + + void sync_game_cursor() { + if (!hardware_cursor_enabled_) return; + const bool visible = !touch_controller.is_enabled() && PythonBoot::CursorVisible() && + !PythonBoot::IsSoftwareCursorVisible(); + if (visible == os_cursor_hidden_) { + if (visible) SDL_ShowCursor(); + else SDL_HideCursor(); + os_cursor_hidden_ = !visible; + } + if (!visible) return; + const int shape = PythonBoot::CursorShape(); + if (shape == current_cursor_shape_) return; + SDL_Cursor* cursor = shape >= 0 && shape < static_cast(game_cursors_.size()) + ? game_cursors_[shape] : nullptr; + if (cursor || fallback_cursor_) SDL_SetCursor(cursor ? cursor : fallback_cursor_); + current_cursor_shape_ = shape; + } +#endif + + explicit VulkanWindow(bool vsync = true, int init_width = 960, int init_height = 640) : vsync_(vsync) { +#ifdef __APPLE__ + if (!std::getenv("VK_ICD_FILENAMES")) { + for (const char* icd : { + "/opt/homebrew/etc/vulkan/icd.d/MoltenVK_icd.json", + "/usr/local/etc/vulkan/icd.d/MoltenVK_icd.json"}) { + if (std::ifstream(icd).good()) { + setenv("VK_ICD_FILENAMES", icd, 0); + break; + } + } + } +#endif + if (!SDL_Init(SDL_INIT_VIDEO)) throw std::runtime_error(SDL_GetError()); + window_ = SDL_CreateWindow( + "Metin2 Native Vulkan Client", + std::max(init_width, 320), + std::max(init_height, 240), + SDL_WINDOW_VULKAN | SDL_WINDOW_RESIZABLE | SDL_WINDOW_HIGH_PIXEL_DENSITY); + if (!window_) throw std::runtime_error(SDL_GetError()); + SDL_StartTextInput(window_); + Uint32 extension_count = 0; + const char* const* sdl_extensions = SDL_Vulkan_GetInstanceExtensions(&extension_count); + if (!sdl_extensions) throw std::runtime_error(SDL_GetError()); + std::vector extensions(sdl_extensions, sdl_extensions + extension_count); + std::uint32_t available_count = 0; + check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, nullptr), "vkEnumerateInstanceExtensionProperties count"); + std::vector available(available_count); + check(vkEnumerateInstanceExtensionProperties(nullptr, &available_count, available.data()), "vkEnumerateInstanceExtensionProperties"); + const bool portability = std::any_of(available.begin(), available.end(), [](const auto& extension) { + return std::strcmp(extension.extensionName, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; + }); + if (portability && std::find_if(extensions.begin(), extensions.end(), [](const char* name) { + return std::strcmp(name, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME) == 0; + }) == extensions.end()) + extensions.push_back(VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME); + VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO}; + app.pApplicationName = "Metin2 native renderer"; + app.apiVersion = VK_API_VERSION_1_1; + VkInstanceCreateInfo instance_info{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO}; + instance_info.flags = portability ? VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR : 0; + instance_info.pApplicationInfo = &app; + instance_info.enabledExtensionCount = static_cast(extensions.size()); + instance_info.ppEnabledExtensionNames = extensions.data(); + check(vkCreateInstance(&instance_info, nullptr, &instance_), "vkCreateInstance"); + if (!SDL_Vulkan_CreateSurface(window_, instance_, nullptr, &surface_)) throw std::runtime_error(SDL_GetError()); + select_device(); + VkCommandPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; + pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; + pool_info.queueFamilyIndex = queue_family_; + check(vkCreateCommandPool(device_, &pool_info, nullptr, &pool_), "vkCreateCommandPool"); + VkCommandBufferAllocateInfo alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + alloc.commandPool = pool_; alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &alloc, &command_), "vkAllocateCommandBuffers"); + VkQueryPoolCreateInfo query_info{VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO}; + query_info.queryType = VK_QUERY_TYPE_TIMESTAMP; + query_info.queryCount = 2; + check(vkCreateQueryPool(device_, &query_info, nullptr, &query_pool_), "vkCreateQueryPool"); + create_swapchain(); + create_descriptors_and_buffers(); + create_render_pass_and_layout(); + VkSemaphoreCreateInfo semaphore_info{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &acquire_), "vkCreateSemaphore acquire"); + check(vkCreateSemaphore(device_, &semaphore_info, nullptr, &rendered_), "vkCreateSemaphore rendered"); + VkFenceCreateInfo fence_info{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; + check(vkCreateFence(device_, &fence_info, nullptr, &fence_), "vkCreateFence"); + touch_controller.update_screen_size(int(width()), int(height())); + } + + ~VulkanWindow() { + if (hardware_cursor_enabled_) { + SDL_SetCursor(nullptr); + for (SDL_Cursor* cursor : game_cursors_) if (cursor) SDL_DestroyCursor(cursor); + if (fallback_cursor_) SDL_DestroyCursor(fallback_cursor_); + } + if (device_) vkDeviceWaitIdle(device_); + if (bone_mapped_) vkUnmapMemory(device_, bone_memory_); + if (bone_buffer_) vkDestroyBuffer(device_, bone_buffer_, nullptr); + if (bone_memory_) vkFreeMemory(device_, bone_memory_, nullptr); + if (state_mapped_) vkUnmapMemory(device_, state_memory_); + if (state_buffer_) vkDestroyBuffer(device_, state_buffer_, nullptr); + if (state_memory_) vkFreeMemory(device_, state_memory_, nullptr); + if (ui_mapped_) vkUnmapMemory(device_, ui_memory_); + if (ui_buffer_) vkDestroyBuffer(device_, ui_buffer_, nullptr); + if (ui_memory_) vkFreeMemory(device_, ui_memory_, nullptr); + if (query_pool_) vkDestroyQueryPool(device_, query_pool_, nullptr); + if (fence_) vkDestroyFence(device_, fence_, nullptr); + if (acquire_) vkDestroySemaphore(device_, acquire_, nullptr); + if (rendered_) vkDestroySemaphore(device_, rendered_, nullptr); + for (auto& entry : geometries_) release_geometry(entry.second); + for (auto& entry : textures_) release_texture(entry.second); + release_texture(fallback_texture_); + for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); + if (vertex_module_) vkDestroyShaderModule(device_, vertex_module_, nullptr); + if (fragment_module_) vkDestroyShaderModule(device_, fragment_module_, nullptr); + if (layout_) vkDestroyPipelineLayout(device_, layout_, nullptr); + if (descriptor_pool_) vkDestroyDescriptorPool(device_, descriptor_pool_, nullptr); + if (descriptor_layout_) vkDestroyDescriptorSetLayout(device_, descriptor_layout_, nullptr); + if (bone_descriptor_layout_) vkDestroyDescriptorSetLayout(device_, bone_descriptor_layout_, nullptr); + if (sampler_) vkDestroySampler(device_, sampler_, nullptr); + if (clamp_sampler_) vkDestroySampler(device_, clamp_sampler_, nullptr); + if (ui_sampler_) vkDestroySampler(device_, ui_sampler_, nullptr); + for (const auto& entry : state_samplers_) vkDestroySampler(device_, entry.second, nullptr); + for (auto framebuffer : framebuffers_) vkDestroyFramebuffer(device_, framebuffer, nullptr); + if (pass_) vkDestroyRenderPass(device_, pass_, nullptr); + if (depth_view_) vkDestroyImageView(device_, depth_view_, nullptr); + if (depth_image_) vkDestroyImage(device_, depth_image_, nullptr); + if (depth_memory_) vkFreeMemory(device_, depth_memory_, nullptr); + for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); + if (swapchain_) vkDestroySwapchainKHR(device_, swapchain_, nullptr); + if (pool_) vkDestroyCommandPool(device_, pool_, nullptr); + if (device_) vkDestroyDevice(device_, nullptr); + if (surface_) vkDestroySurfaceKHR(instance_, surface_, nullptr); + if (instance_) vkDestroyInstance(instance_, nullptr); + if (window_) { + SDL_StopTextInput(window_); + SDL_DestroyWindow(window_); + } + SDL_Quit(); + } + + std::uint32_t width() const { return extent_.width; } + std::uint32_t height() const { return extent_.height; } + std::uint32_t logical_width() const { + int w = 0, h = 0; + if (window_) SDL_GetWindowSize(window_, &w, &h); + return w > 0 ? static_cast(w) : extent_.width; + } + std::uint32_t logical_height() const { + int w = 0, h = 0; + if (window_) SDL_GetWindowSize(window_, &w, &h); + return h > 0 ? static_cast(h) : extent_.height; + } + + TouchController touch_controller; + + static int sdl_scancode_to_dik(SDL_Scancode sc) { + switch (sc) { + case SDL_SCANCODE_ESCAPE: return 0x01; + case SDL_SCANCODE_1: return 0x02; + case SDL_SCANCODE_2: return 0x03; + case SDL_SCANCODE_3: return 0x04; + case SDL_SCANCODE_4: return 0x05; + case SDL_SCANCODE_5: return 0x06; + case SDL_SCANCODE_6: return 0x07; + case SDL_SCANCODE_7: return 0x08; + case SDL_SCANCODE_8: return 0x09; + case SDL_SCANCODE_9: return 0x0A; + case SDL_SCANCODE_0: return 0x0B; + case SDL_SCANCODE_MINUS: return 0x0C; + case SDL_SCANCODE_EQUALS: return 0x0D; + case SDL_SCANCODE_BACKSPACE: return 0x0E; + case SDL_SCANCODE_TAB: return 0x0F; + case SDL_SCANCODE_Q: return 0x10; + case SDL_SCANCODE_W: return 0x11; + case SDL_SCANCODE_E: return 0x12; + case SDL_SCANCODE_R: return 0x13; + case SDL_SCANCODE_T: return 0x14; + case SDL_SCANCODE_Y: return 0x15; + case SDL_SCANCODE_U: return 0x16; + case SDL_SCANCODE_I: return 0x17; + case SDL_SCANCODE_O: return 0x18; + case SDL_SCANCODE_P: return 0x19; + case SDL_SCANCODE_LEFTBRACKET: return 0x1A; + case SDL_SCANCODE_RIGHTBRACKET: return 0x1B; + case SDL_SCANCODE_RETURN: return 0x1C; + case SDL_SCANCODE_LCTRL: return 0x1D; + case SDL_SCANCODE_A: return 0x1E; + case SDL_SCANCODE_S: return 0x1F; + case SDL_SCANCODE_D: return 0x20; + case SDL_SCANCODE_F: return 0x21; + case SDL_SCANCODE_G: return 0x22; + case SDL_SCANCODE_H: return 0x23; + case SDL_SCANCODE_J: return 0x24; + case SDL_SCANCODE_K: return 0x25; + case SDL_SCANCODE_L: return 0x26; + case SDL_SCANCODE_SEMICOLON: return 0x27; + case SDL_SCANCODE_APOSTROPHE: return 0x28; + case SDL_SCANCODE_GRAVE: return 0x29; + case SDL_SCANCODE_LSHIFT: return 0x2A; + case SDL_SCANCODE_BACKSLASH: return 0x2B; + case SDL_SCANCODE_Z: return 0x2C; + case SDL_SCANCODE_X: return 0x2D; + case SDL_SCANCODE_C: return 0x2E; + case SDL_SCANCODE_V: return 0x2F; + case SDL_SCANCODE_B: return 0x30; + case SDL_SCANCODE_N: return 0x31; + case SDL_SCANCODE_M: return 0x32; + case SDL_SCANCODE_COMMA: return 0x33; + case SDL_SCANCODE_PERIOD: return 0x34; + case SDL_SCANCODE_SLASH: return 0x35; + case SDL_SCANCODE_RSHIFT: return 0x36; + case SDL_SCANCODE_KP_MULTIPLY: return 0x37; + case SDL_SCANCODE_LALT: return 0x38; + case SDL_SCANCODE_SPACE: return 0x39; + case SDL_SCANCODE_CAPSLOCK: return 0x3A; + case SDL_SCANCODE_F1: return 0x3B; + case SDL_SCANCODE_F2: return 0x3C; + case SDL_SCANCODE_F3: return 0x3D; + case SDL_SCANCODE_F4: return 0x3E; + case SDL_SCANCODE_F5: return 0x3F; + case SDL_SCANCODE_F6: return 0x40; + case SDL_SCANCODE_F7: return 0x41; + case SDL_SCANCODE_F8: return 0x42; + case SDL_SCANCODE_F9: return 0x43; + case SDL_SCANCODE_F10: return 0x44; + case SDL_SCANCODE_NUMLOCKCLEAR: return 0x45; + case SDL_SCANCODE_SCROLLLOCK: return 0x46; + case SDL_SCANCODE_KP_7: return 0x47; + case SDL_SCANCODE_KP_8: return 0x48; + case SDL_SCANCODE_KP_9: return 0x49; + case SDL_SCANCODE_KP_MINUS: return 0x4A; + case SDL_SCANCODE_KP_4: return 0x4B; + case SDL_SCANCODE_KP_5: return 0x4C; + case SDL_SCANCODE_KP_6: return 0x4D; + case SDL_SCANCODE_KP_PLUS: return 0x4E; + case SDL_SCANCODE_KP_1: return 0x4F; + case SDL_SCANCODE_KP_2: return 0x50; + case SDL_SCANCODE_KP_3: return 0x51; + case SDL_SCANCODE_KP_0: return 0x52; + case SDL_SCANCODE_KP_PERIOD: return 0x53; + case SDL_SCANCODE_F11: return 0x57; + case SDL_SCANCODE_F12: return 0x58; + case SDL_SCANCODE_KP_ENTER: return 0x9C; + case SDL_SCANCODE_RCTRL: return 0x9D; + case SDL_SCANCODE_KP_DIVIDE: return 0xB5; + case SDL_SCANCODE_RALT: return 0xB8; + case SDL_SCANCODE_HOME: return 0xC7; + case SDL_SCANCODE_UP: return 0xC8; + case SDL_SCANCODE_PAGEUP: return 0xC9; + case SDL_SCANCODE_LEFT: return 0xCB; + case SDL_SCANCODE_RIGHT: return 0xCD; + case SDL_SCANCODE_END: return 0xCF; + case SDL_SCANCODE_DOWN: return 0xD0; + case SDL_SCANCODE_PAGEDOWN: return 0xD1; + case SDL_SCANCODE_INSERT: return 0xD2; + case SDL_SCANCODE_DELETE: return 0xD3; + default: return 0; + } + } + + static int sdl_scancode_to_vk(SDL_Scancode sc) { + switch (sc) { + case SDL_SCANCODE_BACKSPACE: return 0x08; + case SDL_SCANCODE_TAB: return 0x09; + case SDL_SCANCODE_RETURN: + case SDL_SCANCODE_KP_ENTER: return 0x0D; + case SDL_SCANCODE_ESCAPE: return 0x1B; + case SDL_SCANCODE_SPACE: return 0x20; + case SDL_SCANCODE_PAGEUP: return 0x21; + case SDL_SCANCODE_PAGEDOWN: return 0x22; + case SDL_SCANCODE_END: return 0x23; + case SDL_SCANCODE_HOME: return 0x24; + case SDL_SCANCODE_LEFT: return 0x25; + case SDL_SCANCODE_UP: return 0x26; + case SDL_SCANCODE_RIGHT: return 0x27; + case SDL_SCANCODE_DOWN: return 0x28; + case SDL_SCANCODE_INSERT: return 0x2D; + case SDL_SCANCODE_DELETE: return 0x2E; + case SDL_SCANCODE_F1: return 0x70; + case SDL_SCANCODE_F2: return 0x71; + case SDL_SCANCODE_F3: return 0x72; + case SDL_SCANCODE_F4: return 0x73; + case SDL_SCANCODE_F5: return 0x74; + case SDL_SCANCODE_F6: return 0x75; + case SDL_SCANCODE_F7: return 0x76; + case SDL_SCANCODE_F8: return 0x77; + case SDL_SCANCODE_F9: return 0x78; + case SDL_SCANCODE_F10: return 0x79; + case SDL_SCANCODE_F11: return 0x7A; + case SDL_SCANCODE_F12: return 0x7B; + default: return 0; + } + } + + bool poll(bool forward_to_live_client = false) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + auto map_mouse = [&](float wx, float wy, int& ox, int& oy) { + int win_w = 0, win_h = 0; + SDL_GetWindowSize(window_, &win_w, &win_h); + unsigned ui_w = extent_.width ? extent_.width : 960; + unsigned ui_h = extent_.height ? extent_.height : 640; + UIRenderGetSize(&ui_w, &ui_h); + if (win_w > 0 && win_h > 0 && ui_w > 0 && ui_h > 0) { + ox = static_cast(std::lround(wx * float(ui_w) / float(win_w))); + oy = static_cast(std::lround(wy * float(ui_h) / float(win_h))); + } else { + ox = static_cast(std::lround(wx)); + oy = static_cast(std::lround(wy)); + } + }; + std::optional> pending_mouse_move; + auto flush_mouse_move = [&] { + if (!pending_mouse_move) return; + const auto [mx, my] = *pending_mouse_move; + pending_mouse_move.reset(); + last_mx_ = mx; + last_my_ = my; + if (touch_controller.is_enabled() && touch_controller.on_mouse_motion(mx, my)) return; + PythonBoot::UIMouseMove(mx, my); + }; +#endif + SDL_Event event; + while (SDL_PollEvent(&event)) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && event.type == SDL_EVENT_MOUSE_MOTION) { + int mx = 0, my = 0; + map_mouse(event.motion.x, event.motion.y, mx, my); + pending_mouse_move = std::pair{mx, my}; + continue; + } + // Preserve event order at button/key/focus boundaries while collapsing + // high-frequency motion into the most recent position. + flush_mouse_move(); +#endif + if (event.type == SDL_EVENT_QUIT) return false; + if (event.type == SDL_EVENT_WINDOW_RESIZED || event.type == SDL_EVENT_WINDOW_PIXEL_SIZE_CHANGED) { + int ew = 0, eh = 0; + SDL_GetWindowSize(window_, &ew, &eh); + touch_controller.update_screen_size(ew, eh); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && ew > 0 && eh > 0) { + PythonBoot::SetUISize(ew, eh); + } +#endif + recreate_swapchain(); + continue; + } + if (event.type == SDL_EVENT_FINGER_DOWN) { + touch_controller.on_finger_down(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } else if (event.type == SDL_EVENT_FINGER_MOTION) { + touch_controller.on_finger_motion(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } else if (event.type == SDL_EVENT_FINGER_UP || event.type == SDL_EVENT_FINGER_CANCELED) { + touch_controller.on_finger_up(event.tfinger.fingerID, event.tfinger.x, event.tfinger.y); + continue; + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client) { + if (!hardware_cursor_enabled_ && + (event.type == SDL_EVENT_WINDOW_MOUSE_ENTER || event.type == SDL_EVENT_WINDOW_FOCUS_GAINED)) { + std::printf(">>> SDL MOUSE ENTER/FOCUS GAINED\n"); + SDL_HideCursor(); + os_cursor_hidden_ = true; + } else if (event.type == SDL_EVENT_WINDOW_FOCUS_LOST) { + SDL_CaptureMouse(false); + PythonBoot::UIMouseButton(2, false, last_mx_, last_my_); + PythonBoot::UIMouseButton(3, false, last_mx_, last_my_); + PythonBoot::UIMouseButton(1, false, last_mx_, last_my_); + } + if (event.type == SDL_EVENT_MOUSE_BUTTON_DOWN || event.type == SDL_EVENT_MOUSE_BUTTON_UP) { + const bool pressed = event.type == SDL_EVENT_MOUSE_BUTTON_DOWN; + int btn = 0; + if (event.button.button == SDL_BUTTON_LEFT) btn = 1; + else if (event.button.button == SDL_BUTTON_RIGHT) btn = 2; + else if (event.button.button == SDL_BUTTON_MIDDLE) btn = 3; + if (btn) { + int mx = 0, my = 0; + map_mouse(event.button.x, event.button.y, mx, my); + last_mx_ = mx; + last_my_ = my; + if (touch_controller.is_enabled()) { + if (touch_controller.on_mouse_button(btn, pressed, mx, my)) { + continue; + } + } + SDL_CaptureMouse(SDL_GetMouseState(nullptr, nullptr) != 0); + PythonBoot::UIMouseButton(btn, pressed, mx, my); + } + } else if (event.type == SDL_EVENT_MOUSE_WHEEL) { + PythonBoot::UIMouseWheel(int(event.wheel.y * 120.0f)); + } else if (event.type == SDL_EVENT_KEY_DOWN || event.type == SDL_EVENT_KEY_UP) { + const bool pressed = event.type == SDL_EVENT_KEY_DOWN; + if (pressed) { + if (const int vk = sdl_scancode_to_vk(event.key.scancode)) + PythonBoot::UIIMEKeyDown(vk); + if (event.key.scancode == SDL_SCANCODE_BACKSPACE) PythonBoot::UIChar(8); + else if (event.key.scancode == SDL_SCANCODE_TAB) PythonBoot::UIChar(9); + else if (event.key.scancode == SDL_SCANCODE_RETURN || event.key.scancode == SDL_SCANCODE_KP_ENTER) + PythonBoot::UIChar(13); + else if (event.key.scancode == SDL_SCANCODE_ESCAPE) PythonBoot::UIChar(27); + } + if (!event.key.repeat) { + if (const int dik = sdl_scancode_to_dik(event.key.scancode)) + PythonBoot::UIKey(dik, pressed); + } + } else if (event.type == SDL_EVENT_TEXT_INPUT && event.text.text) { + const auto* s = reinterpret_cast(event.text.text); + while (*s) { + unsigned cp = 0; + if (*s < 0x80u) { + cp = *s++; + } else if ((*s & 0xE0u) == 0xC0u && (s[1] & 0xC0u) == 0x80u) { + cp = ((*s & 0x1Fu) << 6) | (s[1] & 0x3Fu); + s += 2; + } else if ((*s & 0xF0u) == 0xE0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u) { + cp = ((*s & 0x0Fu) << 12) | ((s[1] & 0x3Fu) << 6) | (s[2] & 0x3Fu); + s += 3; + } else if ((*s & 0xF8u) == 0xF0u && (s[1] & 0xC0u) == 0x80u && (s[2] & 0xC0u) == 0x80u && (s[3] & 0xC0u) == 0x80u) { + cp = ((*s & 0x07u) << 18) | ((s[1] & 0x3Fu) << 12) | ((s[2] & 0x3Fu) << 6) | (s[3] & 0x3Fu); + s += 4; + } else { + ++s; + continue; + } + if (cp >= 32u && cp != 127u) + PythonBoot::UIChar(cp); + } + } + } +#else + (void)forward_to_live_client; +#endif + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + flush_mouse_move(); +#endif + touch_controller.update(); +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (forward_to_live_client && !hardware_cursor_enabled_ && !os_cursor_hidden_ && !touch_controller.is_enabled()) { + SDL_HideCursor(); + os_cursor_hidden_ = true; + } +#endif + return true; + } + + void reset_timings() { + collect_pending_gpu_timestamp(); + timings_ = {}; + timed_frames_ = 0; + upload_count_ = 0; + uploaded_bytes_ = 0; + texture_upload_count_ = 0; + texture_uploaded_bytes_ = 0; + } + + void finish_gpu_timings() { + if (!has_pending_query_) return; + check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences finish"); + collect_pending_gpu_timestamp(); + } + + void render( + const std::vector& draws, + const std::unordered_map>& capture_textures, + std::uint32_t ui_width = 960, + std::uint32_t ui_height = 640, + const std::vector& ui_commands = {}) { + if (extent_.width == 0 || extent_.height == 0) return; + const auto start = std::chrono::steady_clock::now(); + check(vkWaitForFences(device_, 1, &fence_, VK_TRUE, UINT64_MAX), "vkWaitForFences"); + collect_pending_gpu_timestamp(); + prune_textures(); + std::uint32_t image_index = 0; + const auto acquired = vkAcquireNextImageKHR(device_, swapchain_, UINT64_MAX, acquire_, VK_NULL_HANDLE, &image_index); + if (acquired == VK_ERROR_OUT_OF_DATE_KHR) { + recreate_swapchain(); + return; + } + if (acquired != VK_SUCCESS && acquired != VK_SUBOPTIMAL_KHR) check(acquired, "vkAcquireNextImageKHR"); + const auto synchronized = std::chrono::steady_clock::now(); + ++frame_number_; + ++timed_frames_; + + std::vector prepared_draws; + prepared_draws.reserve(draws.size()); + std::unordered_set active_keys; + std::size_t vertex_count = 0, index_count = 0, skinned_draw_count = 0; + std::uint32_t bone_cursor = 0; + std::size_t required_bones = 0; + for (const auto& draw : draws) { + if (!draw.bone_matrices.empty()) { + if (draw.bone_matrices.size() % 16) + throw std::runtime_error("invalid bone palette length"); + required_bones += draw.bone_matrices.size() / 16; + } + } + ensure_bone_capacity(required_bones); + + for (const auto& draw : draws) { + if (draw.positions.empty() || draw.indices.empty()) continue; + const auto signature = draw.geometry_key ? draw.geometry_revision : geometry_hash(draw); + const GeometryId id{draw.geometry_key, signature}; + auto [it, inserted] = geometries_.try_emplace(id); + auto& geometry = it->second; + if (inserted) { + upload_geometry(draw, geometry); + } else if (geometry.vertex_count != draw.positions.size() / 3 || geometry.index_count != draw.indices.size()) { + throw std::runtime_error("geometry key/revision reused with a different size"); + } + geometry.last_used_frame = frame_number_; + if (draw.geometry_key) active_keys.insert(draw.geometry_key); + + float skin_offset_encoded = 0.0f; + if (!draw.bone_matrices.empty() && draw.bone_matrices.size() % 16 == 0 && bone_mapped_) { + const auto bone_count = static_cast(draw.bone_matrices.size() / 16); + if (std::size_t(bone_cursor) + bone_count <= bone_capacity_) { + std::memcpy( + bone_mapped_ + std::size_t(bone_cursor) * 16, + draw.bone_matrices.data(), + draw.bone_matrices.size() * sizeof(float)); + skin_offset_encoded = float(bone_cursor + 1u); + bone_cursor += bone_count; + ++skinned_draw_count; + } else throw std::runtime_error("bone palette capacity exceeded"); + } + + const bool is_specular = draw_is_specular_spheremap(draw); + const bool is_alpha = draw.alpha_blend != 0; + std::uint8_t cull = 0; + // D3D8 cull mode names describe the winding to discard. With the + // shader's Y flip, clockwise is Vulkan's front-facing winding. + if (draw.cull_mode == 2) cull = 2; // D3DCULL_CW: discard clockwise + else if (draw.cull_mode == 3) cull = 1; // D3DCULL_CCW: discard counter-clockwise + const std::uint8_t depth = draw.z_enable ? (draw.z_write ? 2 : 1) : 0; + std::uint8_t blend = 0; + if (is_alpha) { + if (draw.alpha_blend != 0 && draw.src_blend >= 1 && draw.src_blend <= 11 && + draw.dest_blend >= 1 && draw.dest_blend <= 11) { + blend = static_cast((draw.src_blend & 0xFu) | ((draw.dest_blend & 0xFu) << 4)); + } else { + blend = 1; + } + } + const auto pipeline = get_pipeline(cull, depth, blend, draw.lines, + static_cast(draw.z_func)); + const bool bind_tex1 = is_specular || (!draw.texture1.empty() && (draw.color_op[1] > 1 || draw.alpha_op[1] > 1)); + const auto descriptor = get_texture_descriptor( + draw.texture0, bind_tex1 ? draw.texture1 : "", capture_textures, draw); + + auto constants = make_push_constants(draw, skin_offset_encoded); + if (draw.pretransformed) { + const float screen_width = float(logical_width()); + const float screen_height = float(logical_height()); + constants.mvp = {2.0f / screen_width, 0, 0, 0, + 0, 2.0f / screen_height, 0, 0, + 0, 0, 1, 0, + -1, -1, 0, 1}; + } + prepared_draws.push_back({ + &geometry, + pipeline, + descriptor, + constants, + make_fixed_function_state(draw)}); + vertex_count += geometry.vertex_count; + index_count += geometry.index_count; + } + for (auto it = geometries_.begin(); it != geometries_.end();) { + const auto age = frame_number_ - it->second.last_used_frame; + const bool obsolete_revision = it->first.key && active_keys.contains(it->first.key); + if (age > 120 || (age > 1 && (it->first.key == 0 || obsolete_revision))) { + release_geometry(it->second); + it = geometries_.erase(it); + } else ++it; + } + + std::vector ui_batches; + std::size_t ui_quad_count = 0; + if (ui_mapped_) { + const uint32_t target_ui_w = ui_width ? ui_width : 960; + const uint32_t target_ui_h = ui_height ? ui_height : 640; + if (touch_controller.is_enabled()) { + touch_controller.update_screen_size(int(target_ui_w), int(target_ui_h)); + std::vector combined_ui = ui_commands; + touch_controller.append_ui_commands(combined_ui); + if (!combined_ui.empty()) { + build_ui_batches(target_ui_w, target_ui_h, combined_ui, capture_textures, ui_batches, ui_quad_count); + } + } else if (!ui_commands.empty()) { + build_ui_batches(target_ui_w, target_ui_h, ui_commands, capture_textures, ui_batches, ui_quad_count); + } + } + + ensure_state_capacity(prepared_draws.size() + ui_batches.size()); + std::size_t state_index = 0; + for (auto& draw : prepared_draws) { + draw.state_offset = static_cast(state_index * state_stride_); + std::memcpy(state_mapped_ + draw.state_offset, &draw.fixed_state, sizeof(FixedFunctionState)); + ++state_index; + } + const FixedFunctionState ui_state{}; + for (auto& batch : ui_batches) { + batch.state_offset = static_cast(state_index * state_stride_); + std::memcpy(state_mapped_ + batch.state_offset, &ui_state, sizeof(FixedFunctionState)); + ++state_index; + } + + const auto prepared = std::chrono::steady_clock::now(); + check(vkResetFences(device_, 1, &fence_), "vkResetFences"); + check(vkResetCommandBuffer(command_, 0), "vkResetCommandBuffer"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + check(vkBeginCommandBuffer(command_, &begin), "vkBeginCommandBuffer"); + vkCmdResetQueryPool(command_, query_pool_, 0, 2); + vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, query_pool_, 0); + + VkClearValue clears[2]{}; + clears[0].color = {{0.035f, 0.05f, 0.075f, 1.0f}}; + clears[1].depthStencil = {1.0f, 0}; + VkRenderPassBeginInfo pass_begin{VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO}; + pass_begin.renderPass = pass_; pass_begin.framebuffer = framebuffers_.at(image_index); + pass_begin.renderArea.extent = extent_; pass_begin.clearValueCount = 2; pass_begin.pClearValues = clears; + vkCmdBeginRenderPass(command_, &pass_begin, VK_SUBPASS_CONTENTS_INLINE); + + VkPipeline current_pipeline = VK_NULL_HANDLE; + VkDescriptorSet current_descriptor = VK_NULL_HANDLE; + + auto record_ui_pass = [&](bool behind_3d) { + bool ui_bound = false; + for (const auto& batch : ui_batches) { + if (batch.behind_3d != behind_3d) continue; + if (!ui_bound) { + const VkDeviceSize v_offset = 0; + vkCmdBindVertexBuffers(command_, 0, 1, &ui_buffer_, &v_offset); + vkCmdBindIndexBuffer(command_, ui_buffer_, ui_vertex_bytes_, VK_INDEX_TYPE_UINT32); + ui_bound = true; + } + if (batch.pipeline != current_pipeline) { + vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, batch.pipeline); + current_pipeline = batch.pipeline; + } + if (batch.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &batch.descriptor, 0, nullptr); + current_descriptor = batch.descriptor; + } + vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &bone_descriptor_set_, 1, &batch.state_offset); + vkCmdPushConstants( + command_, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(batch.constants), + &batch.constants); + vkCmdDrawIndexed(command_, batch.index_count, 1, batch.first_index, 0, 0); + } + }; + + record_ui_pass(true); + + for (const auto& prepared_draw : prepared_draws) { + if (prepared_draw.pipeline != current_pipeline) { + vkCmdBindPipeline(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, prepared_draw.pipeline); + current_pipeline = prepared_draw.pipeline; + } + if (prepared_draw.descriptor != current_descriptor) { + vkCmdBindDescriptorSets( + command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 0, 1, &prepared_draw.descriptor, 0, nullptr); + current_descriptor = prepared_draw.descriptor; + } + vkCmdBindDescriptorSets(command_, VK_PIPELINE_BIND_POINT_GRAPHICS, layout_, 1, 1, + &bone_descriptor_set_, 1, &prepared_draw.state_offset); + const auto& geometry = *prepared_draw.geometry; + const VkDeviceSize vertex_offset = 0; + vkCmdBindVertexBuffers(command_, 0, 1, &geometry.buffer, &vertex_offset); + vkCmdBindIndexBuffer(command_, geometry.buffer, geometry.index_offset, VK_INDEX_TYPE_UINT32); + vkCmdPushConstants( + command_, + layout_, + VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT, + 0, + sizeof(prepared_draw.constants), + &prepared_draw.constants); + vkCmdDrawIndexed(command_, geometry.index_count, 1, 0, 0, 0); + } + + record_ui_pass(false); + + vkCmdEndRenderPass(command_); + vkCmdWriteTimestamp(command_, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, query_pool_, 1); + check(vkEndCommandBuffer(command_), "vkEndCommandBuffer"); + has_pending_query_ = true; + + const VkPipelineStageFlags stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT; + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.waitSemaphoreCount = 1; submit.pWaitSemaphores = &acquire_; submit.pWaitDstStageMask = &stage; + submit.commandBufferCount = 1; submit.pCommandBuffers = &command_; + submit.signalSemaphoreCount = 1; submit.pSignalSemaphores = &rendered_; + check(vkQueueSubmit(queue_, 1, &submit, fence_), "vkQueueSubmit"); + const auto submitted = std::chrono::steady_clock::now(); + VkPresentInfoKHR present{VK_STRUCTURE_TYPE_PRESENT_INFO_KHR}; + present.waitSemaphoreCount = 1; present.pWaitSemaphores = &rendered_; + present.swapchainCount = 1; present.pSwapchains = &swapchain_; present.pImageIndices = &image_index; + const auto presented = vkQueuePresentKHR(queue_, &present); + if (presented == VK_ERROR_OUT_OF_DATE_KHR || presented == VK_SUBOPTIMAL_KHR) { + recreate_swapchain(); + } else if (presented != VK_SUCCESS) { + check(presented, "vkQueuePresentKHR"); + } + const auto presented_at = std::chrono::steady_clock::now(); + const auto prepare_ms = std::chrono::duration(prepared - synchronized).count(); + timings_.sync_ms += std::chrono::duration(synchronized - start).count(); + timings_.prepare_ms += prepare_ms; + if (timed_frames_ > 1) timings_.steady_prepare_ms += prepare_ms; + timings_.submit_ms += std::chrono::duration(submitted - prepared).count(); + timings_.present_ms += std::chrono::duration(presented_at - submitted).count(); + last_draw_count_ = prepared_draws.size(); + last_skinned_draw_count_ = skinned_draw_count; + last_ui_batch_count_ = ui_batches.size(); + last_ui_quad_count_ = ui_quad_count; + last_vertex_count_ = vertex_count; + last_index_count_ = index_count; + } + + std::size_t draw_count() const { return last_draw_count_; } + std::size_t skinned_draw_count() const { return last_skinned_draw_count_; } + std::size_t ui_batch_count() const { return last_ui_batch_count_; } + std::size_t ui_quad_count() const { return last_ui_quad_count_; } + std::size_t vertex_count() const { return last_vertex_count_; } + std::size_t index_count() const { return last_index_count_; } + std::size_t upload_count() const { return upload_count_; } + std::size_t uploaded_bytes() const { return uploaded_bytes_; } + std::size_t texture_upload_count() const { return texture_upload_count_; } + std::size_t texture_uploaded_bytes() const { return texture_uploaded_bytes_; } + const std::string& device_name() const { return device_name_; } + const char* present_mode_name() const { + switch (present_mode_) { + case VK_PRESENT_MODE_IMMEDIATE_KHR: return "IMMEDIATE"; + case VK_PRESENT_MODE_MAILBOX_KHR: return "MAILBOX"; + default: return "FIFO"; + } + } + Timings timings() const { return timings_; } + +private: + void recreate_swapchain() { + int w = 0, h = 0; + SDL_GetWindowSizeInPixels(window_, &w, &h); + if (w <= 0 || h <= 0) return; + vkDeviceWaitIdle(device_); + for (auto& entry : pipelines_) vkDestroyPipeline(device_, entry.second, nullptr); + pipelines_.clear(); + for (auto fb : framebuffers_) vkDestroyFramebuffer(device_, fb, nullptr); + framebuffers_.clear(); + if (depth_view_) { vkDestroyImageView(device_, depth_view_, nullptr); depth_view_ = VK_NULL_HANDLE; } + if (depth_image_) { vkDestroyImage(device_, depth_image_, nullptr); depth_image_ = VK_NULL_HANDLE; } + if (depth_memory_) { vkFreeMemory(device_, depth_memory_, nullptr); depth_memory_ = VK_NULL_HANDLE; } + for (auto view : image_views_) vkDestroyImageView(device_, view, nullptr); + image_views_.clear(); + if (swapchain_) { vkDestroySwapchainKHR(device_, swapchain_, nullptr); swapchain_ = VK_NULL_HANDLE; } + create_swapchain(); + for (auto view : image_views_) { + VkImageView fb_attachments[2] = {view, depth_view_}; + VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + framebuffer.renderPass = pass_; framebuffer.attachmentCount = 2; framebuffer.pAttachments = fb_attachments; + framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; + VkFramebuffer handle = VK_NULL_HANDLE; + check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer recreate"); + framebuffers_.push_back(handle); + } + get_pipeline(0, 2, 0); + } + + void collect_pending_gpu_timestamp() { + if (!has_pending_query_) return; + has_pending_query_ = false; + std::uint64_t timestamps[2] = {}; + const VkResult res = vkGetQueryPoolResults( + device_, query_pool_, 0, 2, sizeof(timestamps), timestamps, sizeof(std::uint64_t), VK_QUERY_RESULT_64_BIT); + if (res == VK_SUCCESS && timestamps[1] >= timestamps[0]) { + const double ns = double(timestamps[1] - timestamps[0]) * double(timestamp_period_ns_); + timings_.gpu_ms += ns * 1e-6; + ++timings_.gpu_samples; + } + } + + void build_ui_batches( + std::uint32_t ui_width, + std::uint32_t ui_height, + const std::vector& commands, + const std::unordered_map>& capture_textures, + std::vector& batches, + std::size_t& out_quad_count) { + ensure_ui_capacity(commands.size()); + auto* vertices = reinterpret_cast(ui_mapped_); + auto* indices = reinterpret_cast(ui_mapped_ + ui_vertex_bytes_); + std::uint32_t v_count = 0; + std::uint32_t i_count = 0; + const float inv_w = 2.0f / float(ui_width); + const float inv_h = 2.0f / float(ui_height); + + for (const auto& cmd : commands) { + if (v_count + 4 > ui_vertex_capacity_ || i_count + 6 > ui_index_capacity_) + throw std::runtime_error("UI command buffer capacity exceeded"); + if (cmd.kind == UIRenderCommand::Text) continue; + + float x1 = cmd.x1, y1 = cmd.y1, x2 = cmd.x2, y2 = cmd.y2; + float su = cmd.su, sv = cmd.sv, eu = cmd.eu, ev = cmd.ev; + const bool has_clip = cmd.clip_x2 > cmd.clip_x1 && cmd.clip_y2 > cmd.clip_y1; + + float qx[4], qy[4], u[4], v[4], mu[4] = {}, mv[4] = {}; + std::array top_color = unpack_argb(cmd.argb); + std::array bot_color = + cmd.kind == UIRenderCommand::GradientBar ? unpack_argb(cmd.end_argb) : top_color; + if (top_color[3] <= 0.001f && bot_color[3] <= 0.001f) continue; + + if (cmd.kind == UIRenderCommand::Line) { + const float dx = x2 - x1, dy = y2 - y1; + const float len = std::sqrt(dx * dx + dy * dy); + if (len < 0.1f) continue; + const float nx = -dy / len * 0.5f; + const float ny = dx / len * 0.5f; + qx[0] = x1 + nx; qy[0] = y1 + ny; + qx[1] = x2 + nx; qy[1] = y2 + ny; + qx[2] = x1 - nx; qy[2] = y1 - ny; + qx[3] = x2 - nx; qy[3] = y2 - ny; + for (int k = 0; k < 4; ++k) { u[k] = 0.0f; v[k] = 0.0f; } + } else if (cmd.kind == UIRenderCommand::Image && cmd.quad) { + const bool axis_aligned = + std::fabs(cmd.qx[0] - cmd.qx[2]) < 1e-3f && + std::fabs(cmd.qx[1] - cmd.qx[3]) < 1e-3f && + std::fabs(cmd.qy[0] - cmd.qy[1]) < 1e-3f && + std::fabs(cmd.qy[2] - cmd.qy[3]) < 1e-3f; + for (int k = 0; k < 4; ++k) { + mu[k] = cmd.mu[k]; + mv[k] = cmd.mv[k]; + } + if (has_clip && axis_aligned) { + const float ox1 = cmd.qx[0], oy1 = cmd.qy[0], ox2 = cmd.qx[3], oy2 = cmd.qy[3]; + const float cx1 = std::max(ox1, cmd.clip_x1); + const float cy1 = std::max(oy1, cmd.clip_y1); + const float cx2 = std::min(ox2, cmd.clip_x2); + const float cy2 = std::min(oy2, cmd.clip_y2); + if (cx2 <= cx1 || cy2 <= cy1) continue; + float msu = cmd.mu[0], meu = cmd.mu[1]; + float msv = cmd.mv[0], mev = cmd.mv[2]; + if (ox2 > ox1) { + const float t0 = (cx1 - ox1) / (ox2 - ox1); + const float t1 = (cx2 - ox1) / (ox2 - ox1); + const float du = eu - su; + su = cmd.su + du * t0; + eu = cmd.su + du * t1; + const float mdu = meu - msu; + msu = cmd.mu[0] + mdu * t0; + meu = cmd.mu[0] + mdu * t1; + } + if (oy2 > oy1) { + const float t0 = (cy1 - oy1) / (oy2 - oy1); + const float t1 = (cy2 - oy1) / (oy2 - oy1); + const float dv = ev - sv; + sv = cmd.sv + dv * t0; + ev = cmd.sv + dv * t1; + const float mdv = mev - msv; + msv = cmd.mv[0] + mdv * t0; + mev = cmd.mv[0] + mdv * t1; + } + qx[0] = cx1; qy[0] = cy1; + qx[1] = cx2; qy[1] = cy1; + qx[2] = cx1; qy[2] = cy2; + qx[3] = cx2; qy[3] = cy2; + mu[0] = msu; mv[0] = msv; + mu[1] = meu; mv[1] = msv; + mu[2] = msu; mv[2] = mev; + mu[3] = meu; mv[3] = mev; + } else { + if (has_clip && (x2 <= cmd.clip_x1 || x1 >= cmd.clip_x2 || y2 <= cmd.clip_y1 || y1 >= cmd.clip_y2)) + continue; + for (int k = 0; k < 4; ++k) { + qx[k] = cmd.qx[k]; + qy[k] = cmd.qy[k]; + } + } + u[0] = su; v[0] = sv; + u[1] = eu; v[1] = sv; + u[2] = su; v[2] = ev; + u[3] = eu; v[3] = ev; + } else { + if (has_clip) { + x1 = std::max(x1, cmd.clip_x1); + y1 = std::max(y1, cmd.clip_y1); + x2 = std::min(x2, cmd.clip_x2); + y2 = std::min(y2, cmd.clip_y2); + } + if (x2 <= x1 || y2 <= y1) continue; + qx[0] = x1; qy[0] = y1; + qx[1] = x2; qy[1] = y1; + qx[2] = x1; qy[2] = y2; + qx[3] = x2; qy[3] = y2; + u[0] = su; v[0] = sv; + u[1] = eu; v[1] = sv; + u[2] = su; v[2] = ev; + u[3] = eu; v[3] = ev; + } + + const std::string& tex_name = cmd.kind == UIRenderCommand::Image ? cmd.text : std::string(); + const std::string& mask_name = cmd.kind == UIRenderCommand::Image ? cmd.mask : std::string(); + const bool is_masked = !mask_name.empty(); + // CGraphicExpandedImageInstance: SCREEN/COLOR_DODGE use + // INVDESTCOLOR + ONE; MODULATE uses ZERO + SRCCOLOR. + const std::uint8_t blend_mode = cmd.kind == UIRenderCommand::Image + ? ((cmd.blend == 1 || cmd.blend == 2) ? 0x28 : (cmd.blend == 3 ? 0x31 : 1)) + : 1; + const VkPipeline pipeline = get_pipeline(0, 0, blend_mode); + const VkDescriptorSet descriptor = get_ui_texture_descriptor(tex_name, mask_name, capture_textures); + + for (int k = 0; k < 4; ++k) { + Vertex& vert = vertices[v_count + k]; + vert.position[0] = qx[k] * inv_w - 1.0f; + vert.position[1] = 1.0f - qy[k] * inv_h; + vert.position[2] = 0.0f; + vert.normal[0] = 0.0f; vert.normal[1] = 0.0f; vert.normal[2] = 1.0f; + vert.uv[0] = u[k]; vert.uv[1] = v[k]; + const auto& col = (k < 2) ? top_color : bot_color; + vert.color[0] = col[0]; vert.color[1] = col[1]; vert.color[2] = col[2]; vert.color[3] = col[3]; + vert.joints[0] = vert.joints[1] = vert.joints[2] = vert.joints[3] = 0; + vert.weights[0] = 1.0f; vert.weights[1] = vert.weights[2] = vert.weights[3] = 0.0f; + vert.mask_uv[0] = mu[k]; vert.mask_uv[1] = mv[k]; + vert.rhw = 1.0f; + } + indices[i_count + 0] = v_count + 0; + indices[i_count + 1] = v_count + 1; + indices[i_count + 2] = v_count + 2; + indices[i_count + 3] = v_count + 2; + indices[i_count + 4] = v_count + 1; + indices[i_count + 3 + 2] = v_count + 3; + + const float mask_flag = is_masked ? -1.0f : 0.0f; + if (!batches.empty() && + batches.back().behind_3d == cmd.behind_3d && + batches.back().pipeline == pipeline && + batches.back().descriptor == descriptor && + batches.back().constants.light_dir[3] == mask_flag) { + batches.back().index_count += 6; + } else { + UiBatch batch{}; + batch.first_index = i_count; + batch.index_count = 6; + batch.pipeline = pipeline; + batch.descriptor = descriptor; + batch.behind_3d = cmd.behind_3d; + batch.constants.mvp = kIdentityMatrix; + batch.constants.tint_color = {1.0f, 1.0f, 1.0f, 1.0f}; + batch.constants.ambient_emissive = {1.0f, 1.0f, 1.0f, -1.0f}; + batch.constants.light_dir = {0.0f, 0.0f, 1.0f, mask_flag}; + batch.constants.light_diffuse = {0.0f, 0.0f, 0.0f, 0.0f}; + batches.push_back(batch); + } + v_count += 4; + i_count += 6; + ++out_quad_count; + } + } + + std::uint32_t find_memory_type(std::uint32_t type_bits, VkMemoryPropertyFlags flags) const { + VkPhysicalDeviceMemoryProperties properties{}; + vkGetPhysicalDeviceMemoryProperties(physical_, &properties); + for (std::uint32_t i = 0; i < properties.memoryTypeCount; ++i) + if ((type_bits & (1u << i)) && (properties.memoryTypes[i].propertyFlags & flags) == flags) + return i; + throw std::runtime_error("no matching Vulkan memory type"); + } + + void select_device() { + std::uint32_t count = 0; + check(vkEnumeratePhysicalDevices(instance_, &count, nullptr), "vkEnumeratePhysicalDevices count"); + if (!count) throw std::runtime_error("no Vulkan physical device; set VK_ICD_FILENAMES for MoltenVK"); + std::vector candidates(count); + check(vkEnumeratePhysicalDevices(instance_, &count, candidates.data()), "vkEnumeratePhysicalDevices"); + for (auto candidate : candidates) { + std::uint32_t family_count = 0; + vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, nullptr); + std::vector families(family_count); + vkGetPhysicalDeviceQueueFamilyProperties(candidate, &family_count, families.data()); + for (std::uint32_t i = 0; i < family_count; ++i) { + VkBool32 present = VK_FALSE; + check(vkGetPhysicalDeviceSurfaceSupportKHR(candidate, i, surface_, &present), "surface support"); + if ((families[i].queueFlags & VK_QUEUE_GRAPHICS_BIT) && present) { + physical_ = candidate; queue_family_ = i; break; + } + } + if (physical_) break; + } + if (!physical_) throw std::runtime_error("no graphics/present queue family"); + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + device_name_ = properties.deviceName; + timestamp_period_ns_ = properties.limits.timestampPeriod; + VkPhysicalDeviceFeatures supported_features{}; + vkGetPhysicalDeviceFeatures(physical_, &supported_features); + VkPhysicalDeviceFeatures enabled_features{}; + if (supported_features.samplerAnisotropy) { + enabled_features.samplerAnisotropy = VK_TRUE; + max_anisotropy_ = std::min(16.0f, properties.limits.maxSamplerAnisotropy); + } + std::uint32_t extension_count = 0; + check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, nullptr), "device extensions count"); + std::vector available(extension_count); + check(vkEnumerateDeviceExtensionProperties(physical_, nullptr, &extension_count, available.data()), "device extensions"); + std::vector extensions{VK_KHR_SWAPCHAIN_EXTENSION_NAME}; + constexpr const char* portability_subset = "VK_KHR_portability_subset"; + for (const auto& extension : available) + if (std::strcmp(extension.extensionName, portability_subset) == 0) + extensions.push_back(portability_subset); + const float priority = 1.0f; + VkDeviceQueueCreateInfo queue_info{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; + queue_info.queueFamilyIndex = queue_family_; queue_info.queueCount = 1; queue_info.pQueuePriorities = &priority; + VkDeviceCreateInfo device_info{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; + device_info.queueCreateInfoCount = 1; device_info.pQueueCreateInfos = &queue_info; + device_info.enabledExtensionCount = static_cast(extensions.size()); + device_info.ppEnabledExtensionNames = extensions.data(); + device_info.pEnabledFeatures = &enabled_features; + check(vkCreateDevice(physical_, &device_info, nullptr, &device_), "vkCreateDevice"); + vkGetDeviceQueue(device_, queue_family_, 0, &queue_); + } + + void create_swapchain() { + VkSurfaceCapabilitiesKHR capabilities{}; + check(vkGetPhysicalDeviceSurfaceCapabilitiesKHR(physical_, surface_, &capabilities), "surface capabilities"); + std::uint32_t count = 0; + check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, nullptr), "surface formats count"); + if (!count) throw std::runtime_error("surface has no formats"); + std::vector formats(count); + check(vkGetPhysicalDeviceSurfaceFormatsKHR(physical_, surface_, &count, formats.data()), "surface formats"); + VkSurfaceFormatKHR format = formats.front(); + for (auto candidate : formats) if (candidate.format == VK_FORMAT_B8G8R8A8_UNORM) format = candidate; + swapchain_format_ = format.format; + int width = 0, height = 0; + SDL_GetWindowSizeInPixels(window_, &width, &height); + if (capabilities.currentExtent.width != UINT32_MAX) extent_ = capabilities.currentExtent; + else { + extent_.width = std::clamp(std::uint32_t(width), capabilities.minImageExtent.width, capabilities.maxImageExtent.width); + extent_.height = std::clamp(std::uint32_t(height), capabilities.minImageExtent.height, capabilities.maxImageExtent.height); + } + + present_mode_ = VK_PRESENT_MODE_FIFO_KHR; + if (!vsync_) { + std::uint32_t mode_count = 0; + check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, nullptr), "present modes count"); + std::vector modes(mode_count); + if (mode_count) + check(vkGetPhysicalDeviceSurfacePresentModesKHR(physical_, surface_, &mode_count, modes.data()), "present modes"); + if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_IMMEDIATE_KHR) != modes.end()) + present_mode_ = VK_PRESENT_MODE_IMMEDIATE_KHR; + else if (std::find(modes.begin(), modes.end(), VK_PRESENT_MODE_MAILBOX_KHR) != modes.end()) + present_mode_ = VK_PRESENT_MODE_MAILBOX_KHR; + } + + VkSwapchainCreateInfoKHR info{VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR}; + info.surface = surface_; + info.minImageCount = std::min(capabilities.minImageCount + 1, capabilities.maxImageCount ? capabilities.maxImageCount : UINT32_MAX); + info.imageFormat = format.format; info.imageColorSpace = format.colorSpace; info.imageExtent = extent_; + info.imageArrayLayers = 1; info.imageUsage = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT; + info.imageSharingMode = VK_SHARING_MODE_EXCLUSIVE; info.preTransform = capabilities.currentTransform; + info.compositeAlpha = VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR; info.presentMode = present_mode_; + info.clipped = VK_TRUE; + check(vkCreateSwapchainKHR(device_, &info, nullptr, &swapchain_), "vkCreateSwapchainKHR"); + check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, nullptr), "swapchain images count"); + std::vector images(count); + check(vkGetSwapchainImagesKHR(device_, swapchain_, &count, images.data()), "swapchain images"); + for (auto image : images) { + VkImageViewCreateInfo view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view.image = image; view.viewType = VK_IMAGE_VIEW_TYPE_2D; view.format = swapchain_format_; + view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + view.subresourceRange.levelCount = 1; view.subresourceRange.layerCount = 1; + VkImageView handle = VK_NULL_HANDLE; + check(vkCreateImageView(device_, &view, nullptr, &handle), "vkCreateImageView"); + image_views_.push_back(handle); + } + VkImageCreateInfo depth_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + depth_info.imageType = VK_IMAGE_TYPE_2D; + depth_info.format = VK_FORMAT_D32_SFLOAT; + depth_info.extent = {extent_.width, extent_.height, 1}; + depth_info.mipLevels = 1; depth_info.arrayLayers = 1; + depth_info.samples = VK_SAMPLE_COUNT_1_BIT; + depth_info.tiling = VK_IMAGE_TILING_OPTIMAL; + depth_info.usage = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT; + depth_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &depth_info, nullptr, &depth_image_), "vkCreateImage depth"); + VkMemoryRequirements depth_reqs{}; + vkGetImageMemoryRequirements(device_, depth_image_, &depth_reqs); + VkMemoryAllocateInfo depth_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + depth_alloc.allocationSize = depth_reqs.size; + depth_alloc.memoryTypeIndex = find_memory_type(depth_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &depth_alloc, nullptr, &depth_memory_), "vkAllocateMemory depth"); + check(vkBindImageMemory(device_, depth_image_, depth_memory_, 0), "vkBindImageMemory depth"); + VkImageViewCreateInfo depth_view{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + depth_view.image = depth_image_; depth_view.viewType = VK_IMAGE_VIEW_TYPE_2D; + depth_view.format = VK_FORMAT_D32_SFLOAT; + depth_view.subresourceRange.aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT; + depth_view.subresourceRange.levelCount = 1; depth_view.subresourceRange.layerCount = 1; + check(vkCreateImageView(device_, &depth_view, nullptr, &depth_view_), "vkCreateImageView depth"); + } + + void create_host_buffer(VkDeviceSize size, VkBufferUsageFlags usage, VkBuffer& buffer, VkDeviceMemory& memory, void** mapped) { + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = size; + info.usage = usage; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &info, nullptr, &buffer), "vkCreateBuffer host"); + VkMemoryRequirements reqs{}; + vkGetBufferMemoryRequirements(device_, buffer, &reqs); + VkMemoryAllocateInfo alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + alloc.allocationSize = reqs.size; + alloc.memoryTypeIndex = find_memory_type( + reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &alloc, nullptr, &memory), "vkAllocateMemory host"); + check(vkBindBufferMemory(device_, buffer, memory, 0), "vkBindBufferMemory host"); + check(vkMapMemory(device_, memory, 0, size, 0, mapped), "vkMapMemory host"); + } + + void create_descriptors_and_buffers() { + VkSamplerCreateInfo sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + sampler_info.magFilter = VK_FILTER_LINEAR; + sampler_info.minFilter = VK_FILTER_LINEAR; + sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_LINEAR; + sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; + sampler_info.minLod = 0.0f; + sampler_info.maxLod = VK_LOD_CLAMP_NONE; + if (max_anisotropy_ > 1.0f) { + sampler_info.anisotropyEnable = VK_TRUE; + sampler_info.maxAnisotropy = max_anisotropy_; + } + check(vkCreateSampler(device_, &sampler_info, nullptr, &sampler_), "vkCreateSampler"); + + VkSamplerCreateInfo clamp_info = sampler_info; + clamp_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + clamp_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + clamp_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + check(vkCreateSampler(device_, &clamp_info, nullptr, &clamp_sampler_), "vkCreateSampler clamp"); + + VkSamplerCreateInfo ui_sampler_info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + ui_sampler_info.magFilter = VK_FILTER_LINEAR; + ui_sampler_info.minFilter = VK_FILTER_LINEAR; + ui_sampler_info.mipmapMode = VK_SAMPLER_MIPMAP_MODE_NEAREST; + ui_sampler_info.addressModeU = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.addressModeV = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.addressModeW = VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + ui_sampler_info.minLod = 0.0f; + ui_sampler_info.maxLod = 0.0f; + ui_sampler_info.anisotropyEnable = VK_FALSE; + ui_sampler_info.maxAnisotropy = 1.0f; + check(vkCreateSampler(device_, &ui_sampler_info, nullptr, &ui_sampler_), "vkCreateSampler ui"); + + VkDescriptorSetLayoutBinding tex_bindings[2]{}; + tex_bindings[0].binding = 0; + tex_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + tex_bindings[0].descriptorCount = 1; + tex_bindings[0].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; + tex_bindings[1].binding = 1; + tex_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + tex_bindings[1].descriptorCount = 1; + tex_bindings[1].stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT; + VkDescriptorSetLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; + layout_info.bindingCount = 2; + layout_info.pBindings = tex_bindings; + check(vkCreateDescriptorSetLayout(device_, &layout_info, nullptr, &descriptor_layout_), "vkCreateDescriptorSetLayout tex"); + + VkDescriptorSetLayoutBinding bone_bindings[2]{}; + bone_bindings[0].binding = 0; + bone_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + bone_bindings[0].descriptorCount = 1; + bone_bindings[0].stageFlags = VK_SHADER_STAGE_VERTEX_BIT; + bone_bindings[1].binding = 1; + bone_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + bone_bindings[1].descriptorCount = 1; + bone_bindings[1].stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; + VkDescriptorSetLayoutCreateInfo bone_layout_info{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO}; + bone_layout_info.bindingCount = 2; + bone_layout_info.pBindings = bone_bindings; + check(vkCreateDescriptorSetLayout(device_, &bone_layout_info, nullptr, &bone_descriptor_layout_), "vkCreateDescriptorSetLayout bone"); + + VkDescriptorPoolSize pool_sizes[3] = { + {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER, 16384}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, 4}, + {VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC, 4}}; + VkDescriptorPoolCreateInfo pool_info{VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO}; + pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; + pool_info.maxSets = 8192; + pool_info.poolSizeCount = 3; + pool_info.pPoolSizes = pool_sizes; + check(vkCreateDescriptorPool(device_, &pool_info, nullptr, &descriptor_pool_), "vkCreateDescriptorPool"); + + const std::uint8_t white_pixel[4] = {255, 255, 255, 255}; + upload_texture(1, 1, white_pixel, fallback_texture_, false, false); + fallback_texture_.descriptor = allocate_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); + fallback_texture_.ui_descriptor = allocate_ui_texture_descriptor_set(fallback_texture_.view, fallback_texture_.view); + + void* bone_raw = nullptr; + create_host_buffer(kBoneBufferBytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, bone_buffer_, bone_memory_, &bone_raw); + bone_mapped_ = static_cast(bone_raw); + bone_capacity_ = kMaxBonesPerFrame; + for (std::size_t i = 0; i < 16; ++i) bone_mapped_[i] = kIdentityMatrix[i]; + + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + const auto alignment = std::max( + 1, properties.limits.minStorageBufferOffsetAlignment); + state_stride_ = ((sizeof(FixedFunctionState) + alignment - 1) / alignment) * alignment; + state_capacity_ = 256; + void* state_raw = nullptr; + create_host_buffer(state_capacity_ * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + state_buffer_, state_memory_, &state_raw); + state_mapped_ = static_cast(state_raw); + + VkDescriptorSetAllocateInfo bone_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + bone_alloc.descriptorPool = descriptor_pool_; + bone_alloc.descriptorSetCount = 1; + bone_alloc.pSetLayouts = &bone_descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &bone_alloc, &bone_descriptor_set_), "vkAllocateDescriptorSets bone"); + + VkDescriptorBufferInfo buffer_infos[2] = { + {bone_buffer_, 0, kBoneBufferBytes}, + {state_buffer_, 0, sizeof(FixedFunctionState)}}; + VkWriteDescriptorSet writes[2]{}; + for (int binding = 0; binding < 2; ++binding) { + writes[binding].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[binding].dstSet = bone_descriptor_set_; + writes[binding].dstBinding = static_cast(binding); + writes[binding].descriptorCount = 1; + writes[binding].descriptorType = binding == 0 + ? VK_DESCRIPTOR_TYPE_STORAGE_BUFFER : VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + writes[binding].pBufferInfo = &buffer_infos[binding]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + + void* ui_raw = nullptr; + create_host_buffer( + kUiBufferBytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, + ui_buffer_, + ui_memory_, + &ui_raw); + ui_mapped_ = static_cast(ui_raw); + ui_vertex_capacity_ = kMaxUiVerticesPerFrame; + ui_index_capacity_ = kMaxUiIndicesPerFrame; + ui_vertex_bytes_ = kUiVertexBytes; + } + + void ensure_bone_capacity(std::size_t required) { + if (required <= bone_capacity_) return; + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(physical_, &properties); + const std::size_t max_bones = properties.limits.maxStorageBufferRange / (16 * sizeof(float)); + if (required > max_bones || required > UINT32_MAX) + throw std::runtime_error("bone palette exceeds device storage-buffer limit"); + std::size_t next = std::min(max_bones, std::max(required, bone_capacity_ * 2)); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(next * 16 * sizeof(float), VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, buffer, memory, &mapped); + vkUnmapMemory(device_, bone_memory_); + vkDestroyBuffer(device_, bone_buffer_, nullptr); + vkFreeMemory(device_, bone_memory_, nullptr); + bone_buffer_ = buffer; + bone_memory_ = memory; + bone_mapped_ = static_cast(mapped); + bone_capacity_ = next; + VkDescriptorBufferInfo info{bone_buffer_, 0, next * 16 * sizeof(float)}; + VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; + write.dstSet = bone_descriptor_set_; + write.dstBinding = 0; + write.descriptorCount = 1; + write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + write.pBufferInfo = &info; + vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); + } + + void ensure_state_capacity(std::size_t required) { + if (required <= state_capacity_) return; + const auto next = std::max(required, state_capacity_ * 2); + if (next > UINT32_MAX / state_stride_) + throw std::runtime_error("fixed-function state buffer exceeds dynamic offset range"); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(next * state_stride_, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, + buffer, memory, &mapped); + vkUnmapMemory(device_, state_memory_); + vkDestroyBuffer(device_, state_buffer_, nullptr); + vkFreeMemory(device_, state_memory_, nullptr); + state_buffer_ = buffer; + state_memory_ = memory; + state_mapped_ = static_cast(mapped); + state_capacity_ = next; + VkDescriptorBufferInfo info{state_buffer_, 0, sizeof(FixedFunctionState)}; + VkWriteDescriptorSet write{VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET}; + write.dstSet = bone_descriptor_set_; + write.dstBinding = 1; + write.descriptorCount = 1; + write.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC; + write.pBufferInfo = &info; + vkUpdateDescriptorSets(device_, 1, &write, 0, nullptr); + } + + void ensure_ui_capacity(std::size_t command_count) { + if (command_count <= ui_vertex_capacity_ / 4 && command_count <= ui_index_capacity_ / 6) return; + if (command_count > 100'000) throw std::runtime_error("UI command count exceeds safety limit"); + const std::size_t quads = std::max(command_count, ui_vertex_capacity_ / 2); + const VkDeviceSize vertex_bytes = VkDeviceSize(quads) * 4 * sizeof(Vertex); + const VkDeviceSize index_bytes = VkDeviceSize(quads) * 6 * sizeof(std::uint32_t); + if (vertex_bytes + index_bytes > 128ull * 1024ull * 1024ull) + throw std::runtime_error("UI geometry exceeds 128 MiB safety limit"); + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + create_host_buffer(vertex_bytes + index_bytes, + VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT, buffer, memory, &mapped); + vkUnmapMemory(device_, ui_memory_); + vkDestroyBuffer(device_, ui_buffer_, nullptr); + vkFreeMemory(device_, ui_memory_, nullptr); + ui_buffer_ = buffer; + ui_memory_ = memory; + ui_mapped_ = static_cast(mapped); + ui_vertex_capacity_ = quads * 4; + ui_index_capacity_ = quads * 6; + ui_vertex_bytes_ = vertex_bytes; + } + + void create_render_pass_and_layout() { + VkAttachmentDescription attachments[2]{}; + attachments[0].format = swapchain_format_; attachments[0].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[0].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[0].storeOp = VK_ATTACHMENT_STORE_OP_STORE; + attachments[0].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; attachments[0].finalLayout = VK_IMAGE_LAYOUT_PRESENT_SRC_KHR; + attachments[1].format = VK_FORMAT_D32_SFLOAT; attachments[1].samples = VK_SAMPLE_COUNT_1_BIT; + attachments[1].loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR; attachments[1].storeOp = VK_ATTACHMENT_STORE_OP_DONT_CARE; + attachments[1].initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + attachments[1].finalLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL; + + VkAttachmentReference color_ref{0, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL}; + VkAttachmentReference depth_ref{1, VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL}; + VkSubpassDescription subpass{}; + subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS; + subpass.colorAttachmentCount = 1; subpass.pColorAttachments = &color_ref; + subpass.pDepthStencilAttachment = &depth_ref; + + VkSubpassDependency dependency{}; + dependency.srcSubpass = VK_SUBPASS_EXTERNAL; dependency.dstSubpass = 0; + dependency.srcStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; + dependency.dstStageMask = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT | VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT; + dependency.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT; + + VkRenderPassCreateInfo pass_info{VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO}; + pass_info.attachmentCount = 2; pass_info.pAttachments = attachments; + pass_info.subpassCount = 1; pass_info.pSubpasses = &subpass; + pass_info.dependencyCount = 1; pass_info.pDependencies = &dependency; + check(vkCreateRenderPass(device_, &pass_info, nullptr, &pass_), "vkCreateRenderPass"); + + for (auto view : image_views_) { + VkImageView fb_attachments[2] = {view, depth_view_}; + VkFramebufferCreateInfo framebuffer{VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO}; + framebuffer.renderPass = pass_; framebuffer.attachmentCount = 2; framebuffer.pAttachments = fb_attachments; + framebuffer.width = extent_.width; framebuffer.height = extent_.height; framebuffer.layers = 1; + VkFramebuffer handle = VK_NULL_HANDLE; + check(vkCreateFramebuffer(device_, &framebuffer, nullptr, &handle), "vkCreateFramebuffer"); + framebuffers_.push_back(handle); + } + + const auto vert = read_spirv("native.vert.spv"); + const auto frag = read_spirv("native.frag.spv"); + auto module = [&](const std::vector& code) { + VkShaderModuleCreateInfo info{VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO}; + info.codeSize = code.size() * sizeof(std::uint32_t); info.pCode = code.data(); + VkShaderModule handle = VK_NULL_HANDLE; + check(vkCreateShaderModule(device_, &info, nullptr, &handle), "vkCreateShaderModule"); + return handle; + }; + vertex_module_ = module(vert); + fragment_module_ = module(frag); + + VkDescriptorSetLayout set_layouts[2] = {descriptor_layout_, bone_descriptor_layout_}; + VkPushConstantRange push_constants{}; + push_constants.stageFlags = VK_SHADER_STAGE_VERTEX_BIT | VK_SHADER_STAGE_FRAGMENT_BIT; + push_constants.size = sizeof(PushConstants); + VkPipelineLayoutCreateInfo layout_info{VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO}; + layout_info.setLayoutCount = 2; layout_info.pSetLayouts = set_layouts; + layout_info.pushConstantRangeCount = 1; layout_info.pPushConstantRanges = &push_constants; + check(vkCreatePipelineLayout(device_, &layout_info, nullptr, &layout_), "vkCreatePipelineLayout"); + + get_pipeline(0, 2, 0); + } + + VkPipeline get_pipeline(std::uint8_t cull, std::uint8_t depth, std::uint8_t blend, + bool lines = false, std::uint8_t z_func = 4) { + const std::uint32_t key = std::uint32_t(cull) | (std::uint32_t(depth) << 8) | + (std::uint32_t(blend) << 16) | (lines ? (1u << 24) : 0u) | + (std::uint32_t(z_func & 0xfu) << 25); + if (auto it = pipelines_.find(key); it != pipelines_.end()) return it->second; + + VkPipelineShaderStageCreateInfo stages[2]{}; + stages[0].sType = stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; + stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT; stages[0].module = vertex_module_; stages[0].pName = "main"; + stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT; stages[1].module = fragment_module_; stages[1].pName = "main"; + + VkVertexInputBindingDescription binding{0, sizeof(Vertex), VK_VERTEX_INPUT_RATE_VERTEX}; + VkVertexInputAttributeDescription attributes[8] = { + {0, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, position)}, + {1, 0, VK_FORMAT_R32G32B32_SFLOAT, offsetof(Vertex, normal)}, + {2, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, uv)}, + {3, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, color)}, + {4, 0, VK_FORMAT_R8G8B8A8_UINT, offsetof(Vertex, joints)}, + {5, 0, VK_FORMAT_R32G32B32A32_SFLOAT, offsetof(Vertex, weights)}, + {6, 0, VK_FORMAT_R32G32_SFLOAT, offsetof(Vertex, mask_uv)}, + {7, 0, VK_FORMAT_R32_SFLOAT, offsetof(Vertex, rhw)}}; + VkPipelineVertexInputStateCreateInfo input{VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO}; + input.vertexBindingDescriptionCount = 1; input.pVertexBindingDescriptions = &binding; + input.vertexAttributeDescriptionCount = 8; input.pVertexAttributeDescriptions = attributes; + + VkPipelineInputAssemblyStateCreateInfo assembly{VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO}; + assembly.topology = lines ? VK_PRIMITIVE_TOPOLOGY_LINE_LIST : VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST; + + VkViewport viewport{0, 0, float(extent_.width), float(extent_.height), 0, 1}; + VkRect2D scissor{{0, 0}, extent_}; + VkPipelineViewportStateCreateInfo viewport_info{VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO}; + viewport_info.viewportCount = 1; viewport_info.pViewports = &viewport; + viewport_info.scissorCount = 1; viewport_info.pScissors = &scissor; + + VkPipelineRasterizationStateCreateInfo raster{VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO}; + raster.polygonMode = VK_POLYGON_MODE_FILL; + raster.cullMode = cull == 1 ? VK_CULL_MODE_BACK_BIT : (cull == 2 ? VK_CULL_MODE_FRONT_BIT : VK_CULL_MODE_NONE); + raster.frontFace = VK_FRONT_FACE_CLOCKWISE; + raster.lineWidth = 1; + + VkPipelineMultisampleStateCreateInfo multisample{VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO}; + multisample.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT; + + VkPipelineDepthStencilStateCreateInfo depth_stencil{VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO}; + depth_stencil.depthTestEnable = depth > 0 ? VK_TRUE : VK_FALSE; + depth_stencil.depthWriteEnable = depth > 1 ? VK_TRUE : VK_FALSE; + switch (z_func) { + case 1: depth_stencil.depthCompareOp = VK_COMPARE_OP_NEVER; break; + case 2: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS; break; + case 3: depth_stencil.depthCompareOp = VK_COMPARE_OP_EQUAL; break; + case 5: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER; break; + case 6: depth_stencil.depthCompareOp = VK_COMPARE_OP_NOT_EQUAL; break; + case 7: depth_stencil.depthCompareOp = VK_COMPARE_OP_GREATER_OR_EQUAL; break; + case 8: depth_stencil.depthCompareOp = VK_COMPARE_OP_ALWAYS; break; + default: depth_stencil.depthCompareOp = VK_COMPARE_OP_LESS_OR_EQUAL; break; + } + + auto d3d_to_vk_blend = [](std::uint8_t d3d, VkBlendFactor fallback) -> VkBlendFactor { + switch (d3d) { + case 1: return VK_BLEND_FACTOR_ZERO; + case 2: return VK_BLEND_FACTOR_ONE; + case 3: return VK_BLEND_FACTOR_SRC_COLOR; + case 4: return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR; + case 5: return VK_BLEND_FACTOR_SRC_ALPHA; + case 6: return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + case 7: return VK_BLEND_FACTOR_DST_ALPHA; + case 8: return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA; + case 9: return VK_BLEND_FACTOR_DST_COLOR; + case 10: return VK_BLEND_FACTOR_ONE_MINUS_DST_COLOR; + case 11: return VK_BLEND_FACTOR_SRC_ALPHA_SATURATE; + default: return fallback; + } + }; + + VkPipelineColorBlendAttachmentState blend_attachment{}; + blend_attachment.colorWriteMask = 0xf; + if (blend > 0) { + blend_attachment.blendEnable = VK_TRUE; + if (blend >= 16) { + blend_attachment.srcColorBlendFactor = + d3d_to_vk_blend(blend & 0xFu, VK_BLEND_FACTOR_SRC_ALPHA); + blend_attachment.dstColorBlendFactor = + d3d_to_vk_blend((blend >> 4) & 0xFu, VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA); + } else { + blend_attachment.srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA; + blend_attachment.dstColorBlendFactor = + blend == 2 ? VK_BLEND_FACTOR_ONE : VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + } + blend_attachment.colorBlendOp = VK_BLEND_OP_ADD; + blend_attachment.srcAlphaBlendFactor = VK_BLEND_FACTOR_ONE; + blend_attachment.dstAlphaBlendFactor = VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA; + blend_attachment.alphaBlendOp = VK_BLEND_OP_ADD; + } + VkPipelineColorBlendStateCreateInfo blending{VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO}; + blending.attachmentCount = 1; blending.pAttachments = &blend_attachment; + + VkGraphicsPipelineCreateInfo pipeline_info{VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO}; + pipeline_info.stageCount = 2; pipeline_info.pStages = stages; + pipeline_info.pVertexInputState = &input; pipeline_info.pInputAssemblyState = &assembly; + pipeline_info.pViewportState = &viewport_info; pipeline_info.pRasterizationState = &raster; + pipeline_info.pMultisampleState = &multisample; pipeline_info.pDepthStencilState = &depth_stencil; + pipeline_info.pColorBlendState = &blending; + pipeline_info.layout = layout_; pipeline_info.renderPass = pass_; + VkPipeline handle = VK_NULL_HANDLE; + check(vkCreateGraphicsPipelines(device_, VK_NULL_HANDLE, 1, &pipeline_info, nullptr, &handle), "vkCreateGraphicsPipelines"); + pipelines_.emplace(key, handle); + return handle; + } + + VkDescriptorSet allocate_texture_descriptor_set(VkImageView view0, VkImageView view1, + VkSampler sampler0 = VK_NULL_HANDLE, + VkSampler sampler1 = VK_NULL_HANDLE) { + VkDescriptorSet set = VK_NULL_HANDLE; + VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + desc_alloc.descriptorPool = descriptor_pool_; + desc_alloc.descriptorSetCount = 1; + desc_alloc.pSetLayouts = &descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets"); + + VkDescriptorImageInfo image_infos[2]{}; + image_infos[0].sampler = sampler0 ? sampler0 : sampler_; + image_infos[0].imageView = view0; + image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + image_infos[1].sampler = sampler1 ? sampler1 : (clamp_sampler_ ? clamp_sampler_ : sampler_); + image_infos[1].imageView = view1; + image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + + VkWriteDescriptorSet writes[2]{}; + for (int b = 0; b < 2; ++b) { + writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[b].dstSet = set; + writes[b].dstBinding = static_cast(b); + writes[b].descriptorCount = 1; + writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + writes[b].pImageInfo = &image_infos[b]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + return set; + } + + VkDescriptorSet allocate_ui_texture_descriptor_set(VkImageView view0, VkImageView view1) { + VkDescriptorSet set = VK_NULL_HANDLE; + VkDescriptorSetAllocateInfo desc_alloc{VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO}; + desc_alloc.descriptorPool = descriptor_pool_; + desc_alloc.descriptorSetCount = 1; + desc_alloc.pSetLayouts = &descriptor_layout_; + check(vkAllocateDescriptorSets(device_, &desc_alloc, &set), "vkAllocateDescriptorSets ui"); + + VkDescriptorImageInfo image_infos[2]{}; + image_infos[0].sampler = ui_sampler_ ? ui_sampler_ : sampler_; + image_infos[0].imageView = view0; + image_infos[0].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + image_infos[1].sampler = ui_sampler_ ? ui_sampler_ : (clamp_sampler_ ? clamp_sampler_ : sampler_); + image_infos[1].imageView = view1; + image_infos[1].imageLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + + VkWriteDescriptorSet writes[2]{}; + for (int b = 0; b < 2; ++b) { + writes[b].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[b].dstSet = set; + writes[b].dstBinding = static_cast(b); + writes[b].descriptorCount = 1; + writes[b].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + writes[b].pImageInfo = &image_infos[b]; + } + vkUpdateDescriptorSets(device_, 2, writes, 0, nullptr); + return set; + } + + GpuTexture& get_gpu_texture( + const std::string& texture_name, + const std::unordered_map>& capture_textures) { + if (texture_name.empty()) return fallback_texture_; + auto [it, inserted] = textures_.try_emplace(texture_name); + it->second.last_used_frame = frame_number_; + if (!inserted && (it->second.view || frame_number_ - it->second.last_decode_attempt_frame < 60)) + return it->second; + if (inserted) { + it->second.descriptor = fallback_texture_.descriptor; + it->second.ui_descriptor = fallback_texture_.ui_descriptor; + } + it->second.last_decode_attempt_frame = frame_number_; + + mtgodot::Image decoded; + if (auto raw = capture_textures.find(texture_name); raw != capture_textures.end() && !raw->second.empty()) { + decoded = decode_texture_bytes(raw->second.data(), raw->second.size()); + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (!decoded.ok()) { + if (texture_name.rfind("mem:", 0) == 0) { + UIMemoryTexture mem_tex; + if (UIRenderMemoryTexture(texture_name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { + const auto mtra = native_draw_capture::encode_raw_argb_as_mtra( + static_cast(mem_tex.width), + static_cast(mem_tex.height), + mem_tex.argb.data()); + decoded = decode_texture_bytes(mtra.data(), mtra.size()); + } + } else { + std::vector pack_bytes; + if (read_live_pack_texture(texture_name, pack_bytes)) + decoded = decode_texture_bytes(pack_bytes.data(), pack_bytes.size()); + } + } +#endif + if (!decoded.ok()) return it->second; + bool is_ui_tex = false; + if (texture_name.rfind("icon/", 0) == 0 || + texture_name.rfind("d:/ymir work/ui/", 0) == 0 || + texture_name.rfind("d:\\ymir work\\ui\\", 0) == 0 || + texture_name.rfind("locale/", 0) == 0 || + texture_name.rfind("locale\\", 0) == 0 || + texture_name.rfind("mem:", 0) == 0 || + (decoded.w <= 128 && decoded.h <= 128)) { + is_ui_tex = true; + } + upload_texture(decoded.w, decoded.h, decoded.rgba.data(), it->second, true, !is_ui_tex); + it->second.descriptor = allocate_texture_descriptor_set(it->second.view, fallback_texture_.view); + it->second.ui_descriptor = allocate_ui_texture_descriptor_set(it->second.view, fallback_texture_.view); + return it->second; + } + + std::uint64_t sampler_state_key(const Render3DDraw& draw, int stage) const { + return (std::uint64_t(draw.address_u[stage] & 15u)) | + (std::uint64_t(draw.address_v[stage] & 15u) << 4) | + (std::uint64_t(draw.min_filter[stage] & 15u) << 8) | + (std::uint64_t(draw.mag_filter[stage] & 15u) << 12) | + (std::uint64_t(draw.mip_filter[stage] & 15u) << 16); + } + + VkSampler sampler_for_draw(const Render3DDraw& draw, int stage) { + const auto key = sampler_state_key(draw, stage); + if (key == 0) return stage == 0 ? sampler_ : clamp_sampler_; + if (auto it = state_samplers_.find(key); it != state_samplers_.end()) return it->second; + auto address_mode = [](std::uint32_t mode) { + switch (mode) { + case 2: return VK_SAMPLER_ADDRESS_MODE_MIRRORED_REPEAT; + case 3: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_EDGE; + case 4: return VK_SAMPLER_ADDRESS_MODE_CLAMP_TO_BORDER; + default: return VK_SAMPLER_ADDRESS_MODE_REPEAT; + } + }; + VkSamplerCreateInfo info{VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO}; + info.addressModeU = address_mode(draw.address_u[stage]); + info.addressModeV = address_mode(draw.address_v[stage]); + info.addressModeW = VK_SAMPLER_ADDRESS_MODE_REPEAT; + info.borderColor = VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE; + info.minFilter = draw.min_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; + info.magFilter = draw.mag_filter[stage] == 1 ? VK_FILTER_NEAREST : VK_FILTER_LINEAR; + info.mipmapMode = draw.mip_filter[stage] == 1 ? VK_SAMPLER_MIPMAP_MODE_NEAREST : VK_SAMPLER_MIPMAP_MODE_LINEAR; + info.minLod = 0.0f; + info.maxLod = draw.mip_filter[stage] == 0 ? 0.0f : VK_LOD_CLAMP_NONE; + if ((draw.min_filter[stage] == 3 || draw.mag_filter[stage] == 3) && max_anisotropy_ > 1.0f) { + info.anisotropyEnable = VK_TRUE; + info.maxAnisotropy = max_anisotropy_; + } + VkSampler sampler = VK_NULL_HANDLE; + check(vkCreateSampler(device_, &info, nullptr, &sampler), "vkCreateSampler D3D state"); + state_samplers_.emplace(key, sampler); + return sampler; + } + + VkDescriptorSet get_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures, + const Render3DDraw& draw) { + const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); + const auto& tex1 = mask_name.empty() ? fallback_texture_ : get_gpu_texture(mask_name, capture_textures); + const std::string pair_key = "3d\n" + texture0_name + "\n" + mask_name + "\n" + + std::to_string(sampler_state_key(draw, 0)) + "\n" + std::to_string(sampler_state_key(draw, 1)); + if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { + it->second.last_used_frame = frame_number_; + return it->second.set; + } + const VkDescriptorSet set = allocate_texture_descriptor_set( + tex0.view ? tex0.view : fallback_texture_.view, + tex1.view ? tex1.view : fallback_texture_.view, + sampler_for_draw(draw, 0), sampler_for_draw(draw, 1)); + paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); + return set; + } + + VkDescriptorSet get_ui_texture_descriptor( + const std::string& texture0_name, + const std::string& mask_name, + const std::unordered_map>& capture_textures) { + const auto& tex0 = get_gpu_texture(texture0_name, capture_textures); + if (mask_name.empty()) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; + const auto& tex1 = get_gpu_texture(mask_name, capture_textures); + if (!tex0.view || !tex1.view) return tex0.ui_descriptor ? tex0.ui_descriptor : tex0.descriptor; + const std::string pair_key = "ui\n" + texture0_name + "\n" + mask_name; + if (auto it = paired_descriptors_.find(pair_key); it != paired_descriptors_.end()) { + it->second.last_used_frame = frame_number_; + return it->second.set; + } + const VkDescriptorSet set = allocate_ui_texture_descriptor_set( + tex0.view ? tex0.view : fallback_texture_.view, + tex1.view ? tex1.view : fallback_texture_.view); + paired_descriptors_.emplace(pair_key, PairedDescriptor{set, frame_number_}); + return set; + } + + void upload_texture( + std::uint32_t width, + std::uint32_t height, + const std::uint8_t* rgba, + GpuTexture& texture, + bool count_upload, + bool generate_mips = true) { + const VkDeviceSize byte_size = VkDeviceSize(width) * height * 4; + const std::uint32_t mip_levels = generate_mips + ? (static_cast(std::floor(std::log2(std::max(width, height)))) + 1u) + : 1u; + VkBuffer staging_buffer = VK_NULL_HANDLE; + VkDeviceMemory staging_memory = VK_NULL_HANDLE; + VkBufferCreateInfo buffer_info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + buffer_info.size = byte_size; + buffer_info.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT; + buffer_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &buffer_info, nullptr, &staging_buffer), "vkCreateBuffer staging"); + VkMemoryRequirements buffer_reqs{}; + vkGetBufferMemoryRequirements(device_, staging_buffer, &buffer_reqs); + VkMemoryAllocateInfo buffer_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + buffer_alloc.allocationSize = buffer_reqs.size; + buffer_alloc.memoryTypeIndex = find_memory_type( + buffer_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &buffer_alloc, nullptr, &staging_memory), "vkAllocateMemory staging"); + check(vkBindBufferMemory(device_, staging_buffer, staging_memory, 0), "vkBindBufferMemory staging"); + void* mapped = nullptr; + check(vkMapMemory(device_, staging_memory, 0, byte_size, 0, &mapped), "vkMapMemory staging"); + std::memcpy(mapped, rgba, byte_size); + vkUnmapMemory(device_, staging_memory); + + VkImageCreateInfo image_info{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + image_info.imageType = VK_IMAGE_TYPE_2D; + image_info.format = VK_FORMAT_R8G8B8A8_UNORM; + image_info.extent = {width, height, 1}; + image_info.mipLevels = mip_levels; + image_info.arrayLayers = 1; + image_info.samples = VK_SAMPLE_COUNT_1_BIT; + image_info.tiling = VK_IMAGE_TILING_OPTIMAL; + image_info.usage = + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT; + image_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateImage(device_, &image_info, nullptr, &texture.image), "vkCreateImage texture"); + VkMemoryRequirements image_reqs{}; + vkGetImageMemoryRequirements(device_, texture.image, &image_reqs); + VkMemoryAllocateInfo image_alloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + image_alloc.allocationSize = image_reqs.size; + image_alloc.memoryTypeIndex = find_memory_type(image_reqs.memoryTypeBits, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + check(vkAllocateMemory(device_, &image_alloc, nullptr, &texture.memory), "vkAllocateMemory texture"); + check(vkBindImageMemory(device_, texture.image, texture.memory, 0), "vkBindImageMemory texture"); + + VkCommandBuffer upload_cmd = VK_NULL_HANDLE; + VkCommandBufferAllocateInfo cmd_alloc{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cmd_alloc.commandPool = pool_; cmd_alloc.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cmd_alloc.commandBufferCount = 1; + check(vkAllocateCommandBuffers(device_, &cmd_alloc, &upload_cmd), "vkAllocateCommandBuffers texture"); + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + check(vkBeginCommandBuffer(upload_cmd, &begin), "vkBeginCommandBuffer texture"); + + VkImageMemoryBarrier to_transfer{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + to_transfer.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + to_transfer.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + to_transfer.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + to_transfer.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_transfer.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_transfer.image = texture.image; + to_transfer.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &to_transfer); + + VkBufferImageCopy region{}; + region.imageSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 0, 1}; + region.imageExtent = {width, height, 1}; + vkCmdCopyBufferToImage(upload_cmd, staging_buffer, texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ®ion); + + std::int32_t mip_w = static_cast(width); + std::int32_t mip_h = static_cast(height); + for (std::uint32_t i = 1; i < mip_levels; ++i) { + VkImageMemoryBarrier barrier{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + barrier.image = texture.image; + barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + barrier.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 1, 0, 1}; + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &barrier); + + const std::int32_t next_w = std::max(1, mip_w / 2); + const std::int32_t next_h = std::max(1, mip_h / 2); + VkImageBlit blit{}; + blit.srcOffsets[0] = {0, 0, 0}; + blit.srcOffsets[1] = {mip_w, mip_h, 1}; + blit.srcSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i - 1, 0, 1}; + blit.dstOffsets[0] = {0, 0, 0}; + blit.dstOffsets[1] = {next_w, next_h, 1}; + blit.dstSubresource = {VK_IMAGE_ASPECT_COLOR_BIT, i, 0, 1}; + vkCmdBlitImage( + upload_cmd, + texture.image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + texture.image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + 1, &blit, VK_FILTER_LINEAR); + + barrier.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + barrier.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + barrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &barrier); + + mip_w = next_w; + mip_h = next_h; + } + + VkImageMemoryBarrier to_shader{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + to_shader.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + to_shader.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + to_shader.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + to_shader.newLayout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL; + to_shader.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_shader.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + to_shader.image = texture.image; + to_shader.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, mip_levels - 1, 1, 0, 1}; + vkCmdPipelineBarrier( + upload_cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &to_shader); + + check(vkEndCommandBuffer(upload_cmd), "vkEndCommandBuffer texture"); + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.commandBufferCount = 1; submit.pCommandBuffers = &upload_cmd; + check(vkQueueSubmit(queue_, 1, &submit, VK_NULL_HANDLE), "vkQueueSubmit texture"); + check(vkQueueWaitIdle(queue_), "vkQueueWaitIdle texture"); + vkFreeCommandBuffers(device_, pool_, 1, &upload_cmd); + vkDestroyBuffer(device_, staging_buffer, nullptr); + vkFreeMemory(device_, staging_memory, nullptr); + + VkImageViewCreateInfo view_info{VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO}; + view_info.image = texture.image; view_info.viewType = VK_IMAGE_VIEW_TYPE_2D; + view_info.format = VK_FORMAT_R8G8B8A8_UNORM; + view_info.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, mip_levels, 0, 1}; + check(vkCreateImageView(device_, &view_info, nullptr, &texture.view), "vkCreateImageView texture"); + + if (count_upload) { + ++texture_upload_count_; + texture_uploaded_bytes_ += byte_size; + } + } + + void release_texture(GpuTexture& texture) { + for (VkDescriptorSet set : {texture.descriptor, texture.ui_descriptor}) + if (set && set != fallback_texture_.descriptor && set != fallback_texture_.ui_descriptor) + vkFreeDescriptorSets(device_, descriptor_pool_, 1, &set); + if (texture.view) vkDestroyImageView(device_, texture.view, nullptr); + if (texture.image) vkDestroyImage(device_, texture.image, nullptr); + if (texture.memory) vkFreeMemory(device_, texture.memory, nullptr); + texture = {}; + } + + void prune_textures() { + constexpr std::uint64_t static_retention_frames = 120; + constexpr std::uint64_t memory_retention_frames = 2; + for (auto it = paired_descriptors_.begin(); it != paired_descriptors_.end();) { + const auto retention = it->first.find("mem:") != std::string::npos + ? memory_retention_frames : static_retention_frames; + if (frame_number_ - it->second.last_used_frame > retention) { + vkFreeDescriptorSets(device_, descriptor_pool_, 1, &it->second.set); + it = paired_descriptors_.erase(it); + } else ++it; + } + for (auto it = textures_.begin(); it != textures_.end();) { + const auto retention = it->first.rfind("mem:", 0) == 0 + ? memory_retention_frames : static_retention_frames; + if (frame_number_ - it->second.last_used_frame > retention) { + release_texture(it->second); + it = textures_.erase(it); + } else ++it; + } + } + + void release_geometry(Geometry& geometry) { + if (geometry.buffer) vkDestroyBuffer(device_, geometry.buffer, nullptr); + if (geometry.memory) vkFreeMemory(device_, geometry.memory, nullptr); + geometry = {}; + } + + void upload_geometry(const Render3DDraw& draw, Geometry& geometry) { + if (draw.positions.size() % 3 || draw.indices.size() % (draw.lines ? 2 : 3)) + throw std::runtime_error("invalid native geometry array size"); + const auto count = draw.positions.size() / 3; + if (count > UINT32_MAX || draw.indices.size() > UINT32_MAX) + throw std::runtime_error("geometry exceeds Vulkan index range"); + const std::uint32_t default_argb = + (!draw.texture0.empty() || draw.lighting != 0) ? 0xffffffffu : 0xff4fbfffu; + const bool has_skin = + draw.bone_indices.size() >= count * 4 && draw.bone_weights.size() >= count * 4; + std::vector vertices(count); + for (std::size_t i = 0; i < count; ++i) { + vertices[i].position[0] = draw.positions[3*i]; + vertices[i].position[1] = draw.positions[3*i+1]; + vertices[i].position[2] = draw.positions[3*i+2]; + vertices[i].rhw = i < draw.rhw.size() ? draw.rhw[i] : 1.0f; + if (3*i + 2 < draw.normals.size()) { + vertices[i].normal[0] = draw.normals[3*i]; + vertices[i].normal[1] = draw.normals[3*i+1]; + vertices[i].normal[2] = draw.normals[3*i+2]; + } else { + vertices[i].normal[0] = 0.0f; + vertices[i].normal[1] = 0.0f; + vertices[i].normal[2] = 1.0f; + } + if (2*i + 1 < draw.uv0.size()) { + vertices[i].uv[0] = draw.uv0[2*i]; + vertices[i].uv[1] = draw.uv0[2*i+1]; + } else { + vertices[i].uv[0] = 0.0f; + vertices[i].uv[1] = 0.0f; + } + const auto argb = i < draw.diffuse.size() ? draw.diffuse[i] : default_argb; + vertices[i].color[0] = float((argb >> 16) & 255) / 255.0f; + vertices[i].color[1] = float((argb >> 8) & 255) / 255.0f; + vertices[i].color[2] = float(argb & 255) / 255.0f; + vertices[i].color[3] = float((argb >> 24) & 255) / 255.0f; + if (has_skin) { + for (int k = 0; k < 4; ++k) { + vertices[i].joints[k] = draw.bone_indices[4*i + k]; + vertices[i].weights[k] = draw.bone_weights[4*i + k]; + } + } else { + vertices[i].joints[0] = vertices[i].joints[1] = vertices[i].joints[2] = vertices[i].joints[3] = 0; + vertices[i].weights[0] = 1.0f; + vertices[i].weights[1] = vertices[i].weights[2] = vertices[i].weights[3] = 0.0f; + } + if (2*i + 1 < draw.uv1.size()) { + vertices[i].mask_uv[0] = draw.uv1[2*i]; + vertices[i].mask_uv[1] = draw.uv1[2*i+1]; + } else { + vertices[i].mask_uv[0] = 0.0f; + vertices[i].mask_uv[1] = 0.0f; + } + } + for (auto index : draw.indices) if (index >= count) throw std::runtime_error("draw index exceeds vertex count"); + geometry.index_offset = vertices.size() * sizeof(Vertex); + const auto bytes = geometry.index_offset + draw.indices.size() * sizeof(std::uint32_t); + if (bytes > 128 * 1024 * 1024) throw std::runtime_error("native geometry buffer exceeds 128 MiB"); + VkBufferCreateInfo info{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + info.size = bytes; info.usage = VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT; + info.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + check(vkCreateBuffer(device_, &info, nullptr, &geometry.buffer), "vkCreateBuffer"); + VkMemoryRequirements requirements{}; + vkGetBufferMemoryRequirements(device_, geometry.buffer, &requirements); + VkMemoryAllocateInfo allocation{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocation.allocationSize = requirements.size; + allocation.memoryTypeIndex = find_memory_type( + requirements.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + check(vkAllocateMemory(device_, &allocation, nullptr, &geometry.memory), "vkAllocateMemory"); + check(vkBindBufferMemory(device_, geometry.buffer, geometry.memory, 0), "vkBindBufferMemory"); + void* mapped = nullptr; + check(vkMapMemory(device_, geometry.memory, 0, bytes, 0, &mapped), "vkMapMemory"); + std::memcpy(mapped, vertices.data(), geometry.index_offset); + std::memcpy(static_cast(mapped) + geometry.index_offset, + draw.indices.data(), draw.indices.size() * sizeof(std::uint32_t)); + vkUnmapMemory(device_, geometry.memory); + geometry.vertex_count = static_cast(count); + geometry.index_count = static_cast(draw.indices.size()); + ++upload_count_; + uploaded_bytes_ += bytes; + } + + bool vsync_ = true; + VkPresentModeKHR present_mode_ = VK_PRESENT_MODE_FIFO_KHR; + float timestamp_period_ns_ = 1.0f; + float max_anisotropy_ = 1.0f; + SDL_Window* window_ = nullptr; + VkInstance instance_ = VK_NULL_HANDLE; + VkSurfaceKHR surface_ = VK_NULL_HANDLE; + VkPhysicalDevice physical_ = VK_NULL_HANDLE; + VkDevice device_ = VK_NULL_HANDLE; + VkQueue queue_ = VK_NULL_HANDLE; + std::uint32_t queue_family_ = 0; + std::string device_name_; + VkSwapchainKHR swapchain_ = VK_NULL_HANDLE; + VkFormat swapchain_format_ = VK_FORMAT_UNDEFINED; + VkExtent2D extent_{}; + std::vector image_views_; + VkImage depth_image_ = VK_NULL_HANDLE; + VkDeviceMemory depth_memory_ = VK_NULL_HANDLE; + VkImageView depth_view_ = VK_NULL_HANDLE; + VkRenderPass pass_ = VK_NULL_HANDLE; + std::vector framebuffers_; + VkSampler sampler_ = VK_NULL_HANDLE; + VkSampler clamp_sampler_ = VK_NULL_HANDLE; + VkSampler ui_sampler_ = VK_NULL_HANDLE; + std::unordered_map state_samplers_; + VkDescriptorSetLayout descriptor_layout_ = VK_NULL_HANDLE; + VkDescriptorSetLayout bone_descriptor_layout_ = VK_NULL_HANDLE; + VkDescriptorPool descriptor_pool_ = VK_NULL_HANDLE; + VkDescriptorSet bone_descriptor_set_ = VK_NULL_HANDLE; + VkBuffer bone_buffer_ = VK_NULL_HANDLE; + VkDeviceMemory bone_memory_ = VK_NULL_HANDLE; + float* bone_mapped_ = nullptr; + std::size_t bone_capacity_ = 0; + VkBuffer state_buffer_ = VK_NULL_HANDLE; + VkDeviceMemory state_memory_ = VK_NULL_HANDLE; + std::uint8_t* state_mapped_ = nullptr; + VkDeviceSize state_stride_ = 0; + std::size_t state_capacity_ = 0; + VkBuffer ui_buffer_ = VK_NULL_HANDLE; + VkDeviceMemory ui_memory_ = VK_NULL_HANDLE; + std::uint8_t* ui_mapped_ = nullptr; + std::size_t ui_vertex_capacity_ = 0, ui_index_capacity_ = 0; + VkDeviceSize ui_vertex_bytes_ = 0; + GpuTexture fallback_texture_{}; + std::unordered_map textures_; + std::unordered_map paired_descriptors_; + VkShaderModule vertex_module_ = VK_NULL_HANDLE; + VkShaderModule fragment_module_ = VK_NULL_HANDLE; + VkPipelineLayout layout_ = VK_NULL_HANDLE; + std::unordered_map pipelines_; + std::unordered_map geometries_; + std::uint64_t frame_number_ = 0; + std::uint64_t timed_frames_ = 0; + std::size_t upload_count_ = 0, uploaded_bytes_ = 0; + std::size_t texture_upload_count_ = 0, texture_uploaded_bytes_ = 0; + VkCommandPool pool_ = VK_NULL_HANDLE; + VkCommandBuffer command_ = VK_NULL_HANDLE; + VkQueryPool query_pool_ = VK_NULL_HANDLE; + bool has_pending_query_ = false; + bool os_cursor_hidden_ = false; + bool hardware_cursor_enabled_ = false; + std::array game_cursors_{}; + SDL_Cursor* fallback_cursor_ = nullptr; + int current_cursor_shape_ = -1; + int last_mx_ = 0, last_my_ = 0; + VkSemaphore acquire_ = VK_NULL_HANDLE, rendered_ = VK_NULL_HANDLE; + VkFence fence_ = VK_NULL_HANDLE; + std::size_t last_draw_count_ = 0, last_skinned_draw_count_ = 0; + std::size_t last_ui_batch_count_ = 0, last_ui_quad_count_ = 0; + std::size_t last_vertex_count_ = 0, last_index_count_ = 0; + Timings timings_{}; +}; + +std::vector make_test_draws(std::size_t count, std::size_t triangles_per_draw) { + std::vector draws; + draws.reserve(count); + for (std::size_t i = 0; i < count; ++i) { + Render3DDraw draw{}; + draw.geometry_key = i + 1; + draw.geometry_revision = 1; + draw.z_enable = 1; + draw.z_write = 1; + for (auto* matrix : {draw.world, draw.view, draw.proj}) + for (int diagonal = 0; diagonal < 4; ++diagonal) matrix[diagonal * 5] = 1.0f; + const float x = (float(i % 8) - 3.5f) * 0.24f; + const float y = (float(i / 8) - 3.5f) * 0.22f; + draw.positions.reserve(triangles_per_draw * 9); + draw.indices.reserve(triangles_per_draw * 3); + draw.diffuse.reserve(triangles_per_draw * 3); + for (std::size_t t = 0; t < triangles_per_draw; ++t) { + const float tx = x + (float(t % 16) - 7.5f) * 0.009f; + const float ty = y + (float(t / 16) - float(triangles_per_draw / 32)) * 0.006f; + const auto base = static_cast(draw.positions.size() / 3); + draw.positions.insert(draw.positions.end(), {tx-0.004f, ty-0.004f, 0.5f, + tx+0.004f, ty-0.004f, 0.5f, + tx, ty+0.004f, 0.5f}); + draw.indices.insert(draw.indices.end(), {base, base+1, base+2}); + draw.diffuse.insert(draw.diffuse.end(), {0xff4fbfff, 0xff4fbfff, 0xff4fbfff}); + } + draws.push_back(std::move(draw)); + } + return draws; +} + +void print_summary(const VulkanWindow& renderer, int completed, double wall_ms, double update_ms = 0.0, + std::vector frame_ms = {}) { + const auto timing = renderer.timings(); + std::sort(frame_ms.begin(), frame_ms.end()); + const auto percentile = [&](double fraction) { + if (frame_ms.empty()) return 0.0; + const auto index = static_cast(std::ceil(fraction * frame_ms.size())) - 1; + return frame_ms[std::min(index, frame_ms.size() - 1)]; + }; + std::cout << "device=" << renderer.device_name() + << " present_mode=" << renderer.present_mode_name() + << " frames=" << completed + << " draws=" << renderer.draw_count() + << " skinned_draws=" << renderer.skinned_draw_count() + << " ui_batches=" << renderer.ui_batch_count() + << " ui_quads=" << renderer.ui_quad_count() + << " vertices=" << renderer.vertex_count() + << " indices=" << renderer.index_count() + << " uploads=" << renderer.upload_count() + << " uploaded_bytes=" << renderer.uploaded_bytes() + << " texture_uploads=" << renderer.texture_upload_count() + << " texture_bytes=" << renderer.texture_uploaded_bytes() + << " mean_frame_ms=" << (completed ? wall_ms / completed : 0.0) + << " p95_frame_ms=" << percentile(0.95) + << " p99_frame_ms=" << percentile(0.99) + << " max_frame_ms=" << percentile(1.0) + << " game_update_ms=" << (completed ? update_ms / completed : 0.0) + << " sync_ms=" << (completed ? timing.sync_ms / completed : 0.0) + << " prepare_ms=" << (completed ? timing.prepare_ms / completed : 0.0) + << " steady_prepare_ms=" << (completed > 1 ? timing.steady_prepare_ms / (completed - 1) : 0.0) + << " submit_ms=" << (completed ? timing.submit_ms / completed : 0.0) + << " present_ms=" << (completed ? timing.present_ms / completed : 0.0) + << " gpu_ms=" << (timing.gpu_samples ? timing.gpu_ms / double(timing.gpu_samples) : 0.0) << '\n'; +} + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +bool py_exec(const std::string& code) { + std::string err; + if (!PythonBoot::RunLine(code.c_str(), &err)) { + std::cerr << "PythonBoot::RunLine failed: " << err << '\n'; + return false; + } + return true; +} + +bool py_eval_true(const char* expr) { + std::string res, err; + return PythonBoot::Evaluate(expr, &res, &err) && (res == "True" || res == "1"); +} + +bool parse_live_server_spec(const std::string& spec, std::string& host, int& auth_port, int& game_port) { + const auto p1 = spec.find(':'); + if (p1 == std::string::npos) return false; + const auto p2 = spec.find(':', p1 + 1); + if (p2 == std::string::npos) return false; + host = spec.substr(0, p1); + auth_port = std::stoi(spec.substr(p1 + 1, p2 - p1 - 1)); + game_port = std::stoi(spec.substr(p2 + 1)); + return !host.empty() && auth_port > 0 && game_port > 0; +} + +int run_live_client( + VulkanWindow& renderer, + const std::string& client_dir, + int frames, + int fake_mobs, + bool gpu_skinning, + bool native_terrain, + bool login_screen, + const std::string& live_server_spec, + const std::string& capture_out) { + if (fake_mobs > 0) { + const std::string mobs_str = std::to_string(fake_mobs); + setenv("MT_FAKE_MOB_COUNT", mobs_str.c_str(), 1); + } + char resolved_client_dir[4096] = {}; + const std::string abs_client_dir = realpath(client_dir.c_str(), resolved_client_dir) ? std::string(resolved_client_dir) : client_dir; + setenv("MT_40250_CLIENT", abs_client_dir.c_str(), 1); + if (chdir(abs_client_dir.c_str()) != 0 || !mtpack40250::initialize(".")) { + throw std::runtime_error("cannot initialize 40250 pack at: " + abs_client_dir); + } + + const bool host_hardware_cursor = !renderer.touch_controller.is_enabled(); + PythonBoot::SetHostHardwareCursorEnabled(host_hardware_cursor); + if (host_hardware_cursor) renderer.enable_game_hardware_cursor(); + + SetNativeTerrainRenderEnabled(native_terrain); + SetGpuSkinningEnabled(gpu_skinning); + NativeAudioEngine audio; + + std::string server_host = "127.0.0.1"; + int auth_port = 0; + int game_port = 0; + const bool use_external_server = !live_server_spec.empty(); + FakeLoginServer server; + if (use_external_server) { + if (!parse_live_server_spec(live_server_spec, server_host, auth_port, game_port)) + throw std::runtime_error("invalid --live-server spec (expected HOST:AUTH_PORT:GAME_PORT): " + live_server_spec); + } else { + if (!server.Start()) throw std::runtime_error("FakeLoginServer failed to start on loopback"); + auth_port = server.AuthPort(); + game_port = server.GamePort(); + } + + std::string error; + const char* env_stdlib = std::getenv("MT_PYTHON_STDLIB"); + std::string stdlib = (env_stdlib && *env_stdlib) ? env_stdlib : bundled_file_path("python27.zip"); +#ifdef __ANDROID__ + if (!env_stdlib || !*env_stdlib) { + std::size_t zip_size = 0; + void* zip = SDL_LoadFile("assets://python27.zip", &zip_size); + if (!zip) zip = SDL_LoadFile("python27.zip", &zip_size); + if (!zip) throw std::runtime_error("cannot load bundled python27.zip"); + char* pref = SDL_GetPrefPath("mtgodot", "native-render"); + if (!pref) { SDL_free(zip); throw std::runtime_error("cannot locate app data directory"); } + stdlib = std::string(pref) + "python27.zip"; + SDL_free(pref); + std::ofstream output(stdlib, std::ios::binary | std::ios::trunc); + output.write(static_cast(zip), static_cast(zip_size)); + SDL_free(zip); + if (!output) throw std::runtime_error("cannot extract bundled python27.zip"); + } +#endif + if (!PythonBoot::Start(stdlib.c_str(), &error)) + throw std::runtime_error("PythonBoot::Start: " + error); + + SetPlatformServerTime(123456789); + PythonBoot::SetUISize(static_cast(renderer.logical_width()), static_cast(renderer.logical_height())); + if (!PythonBoot::RunMainScript("", &error) || !PythonBoot::IsAppLooping()) + throw std::runtime_error("PythonBoot::RunMainScript: " + error); + + const std::unordered_map> empty_textures; + auto pump_until = [&](double max_seconds, int sleep_ms, const auto& cond) -> bool { + const auto deadline = std::chrono::steady_clock::now() + std::chrono::duration(max_seconds); + while (std::chrono::steady_clock::now() < deadline && renderer.poll(true) && PythonBoot::IsAppLooping()) { + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + PythonBoot::UIRender(); + audio.pump(); + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); + if (cond()) return true; + if (sleep_ms > 0) + std::this_thread::sleep_for(std::chrono::milliseconds(sleep_ms)); + } + return cond(); + }; + + if (!py_exec("import __main__, app, networkModule, introLogin, introSelect, introLoading, game\n" + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]")) + throw std::runtime_error("failed to locate networkModule.MainStream"); + + if (!pump_until(20.0, 5, [] { return py_eval_true("isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)"); })) + throw std::runtime_error("timed out waiting for LoginWindow"); + + if (login_screen) { + char conn_cmd[512]; + std::snprintf( + conn_cmd, + sizeof(conn_cmd), + "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" + "_w = _stream.curPhaseWindow\n" + "_w._LoginWindow__OpenServerBoard()", + server_host.c_str(), + game_port, + server_host.c_str(), + auth_port); + if (!py_exec(conn_cmd)) + throw std::runtime_error("failed to configure LoginWindow connection info"); + for (int i = 0; i < 15; ++i) { + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + PythonBoot::UIRender(); + } + } else { + char login_cmd[512]; + std::snprintf( + login_cmd, + sizeof(login_cmd), + "_stream.SetConnectInfo('%s', %d, '%s', %d)\n" + "_w = _stream.curPhaseWindow\n" + "_w._LoginWindow__OpenLoginBoard()\n" + "_w.idEditLine.SetText('%s')\n" + "_w.pwdEditLine.SetText('%s')\n" + "_w._LoginWindow__OnClickLoginButton()", + server_host.c_str(), + game_port, + server_host.c_str(), + auth_port, + FakeLoginServer::kLogin, + FakeLoginServer::kPassword); + if (!py_exec(login_cmd)) + throw std::runtime_error("failed to submit login credentials"); + + if (!pump_until(20.0, 5, [&] { + return (use_external_server || server.Has("game1:login2")) && + py_eval_true("isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)"); + })) + throw std::runtime_error("timed out waiting for SelectCharacterWindow"); + + if (!py_exec("_stream.curPhaseWindow.SelectSlot(0)\n" + "_stream.curPhaseWindow.StartGame()")) + throw std::runtime_error("failed to start game from SelectCharacterWindow"); + + if (!use_external_server) { + if (!pump_until(30.0, 5, [&] { return server.Has("game2:client_version"); })) + throw std::runtime_error("timed out waiting for LoadingWindow (client_version)"); + } + + if (!pump_until(60.0, 0, [&] { + return (use_external_server || server.Has("game2:burst_pong")) && + py_eval_true("isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()"); + })) + throw std::runtime_error("timed out waiting for GameWindow (server error: " + server.Error() + ")"); + + const int warmup_frames = std::max(10, fake_mobs / 2 + 10); + for (int i = 0; i < warmup_frames; ++i) { + if (!renderer.poll(true) || !PythonBoot::IsAppLooping()) break; + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + audio.pump(); + } + } + + // Render one warmup frame to upload initial scene & UI textures/geometries before timed benchmark. + { + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); + } + + if (!capture_out.empty()) { + std::unordered_map> cap_textures; + auto add_tex = [&](const std::string& name) { + if (name.empty() || cap_textures.count(name)) return; + if (name.rfind("mem:", 0) == 0) { + UIMemoryTexture mem_tex; + if (UIRenderMemoryTexture(name, &mem_tex) && mem_tex.width > 0 && mem_tex.height > 0) { + auto mtra = native_draw_capture::encode_raw_argb_as_mtra( + static_cast(mem_tex.width), + static_cast(mem_tex.height), + mem_tex.argb.data()); + if (!mtra.empty()) cap_textures.emplace(name, std::move(mtra)); + } + return; + } + std::vector bytes; + if (read_live_pack_texture(name, bytes)) + cap_textures.emplace(name, std::move(bytes)); + }; + for (const auto& d : Render3DDraws()) { + add_tex(d.texture0); + add_tex(d.texture1); + } + for (const auto& c : UIRenderCommands()) { + if (c.kind == UIRenderCommand::Image) { + add_tex(c.text); + add_tex(c.mask); + } + } + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + std::vector cap_ui = UIRenderCommands(); + if (renderer.touch_controller.is_enabled()) { + renderer.touch_controller.update_screen_size(int(ui_w), int(ui_h)); + renderer.touch_controller.append_ui_commands(cap_ui); + } + native_draw_capture::write(capture_out, Render3DDraws(), cap_textures, ui_w, ui_h, cap_ui); + } + + renderer.reset_timings(); + const auto start = std::chrono::steady_clock::now(); + double update_ms = 0.0; + int completed = 0; + std::vector frame_ms; + if (frames > 0) frame_ms.reserve(static_cast(frames)); + while ((frames == 0 || completed < frames) && renderer.poll(true) && PythonBoot::IsAppLooping()) { + const auto frame_start = std::chrono::steady_clock::now(); + const auto t0 = std::chrono::steady_clock::now(); + PythonBoot::UIUpdate(); + renderer.sync_game_cursor(); + PythonBoot::UIRender(); + audio.pump(); + const auto t1 = std::chrono::steady_clock::now(); + update_ms += std::chrono::duration(t1 - t0).count(); + unsigned ui_w = renderer.width(), ui_h = renderer.height(); + UIRenderGetSize(&ui_w, &ui_h); + renderer.render(Render3DDraws(), empty_textures, ui_w, ui_h, UIRenderCommands()); + frame_ms.push_back(std::chrono::duration( + std::chrono::steady_clock::now() - frame_start).count()); + ++completed; + } + renderer.finish_gpu_timings(); + const auto wall_ms = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); + print_summary(renderer, completed, wall_ms, update_ms, std::move(frame_ms)); + + PythonBoot::Stop(); + if (!use_external_server) + server.Stop(); + return (frames == 0 || completed == frames) ? 0 : 2; +} +#endif + +} // namespace + +int main(int argc, char** argv) { + try { + int frames = 180; + bool frames_specified = false; + bool synthetic_specified = false; + bool interactive = false; + bool login_screen = false; + int init_width = 1280; + int init_height = 800; + std::size_t draw_count = 64; + std::size_t triangles_per_draw = 333; + std::string capture_path; + std::string capture_out; + std::string live_client_dir; + std::string live_server_spec; + int fake_mobs = 64; + bool animate_first_draw = false; + bool animate_bones = false; + bool vsync = true; + bool gpu_skinning = true; + bool native_terrain = true; + bool mobile_mode = false; +#ifdef __ANDROID__ + mobile_mode = true; +#endif + for (int i = 1; i < argc; ++i) { + const std::string arg = argv[i]; + if (arg == "--frames" && i+1 < argc) { + frames = std::stoi(argv[++i]); + frames_specified = true; + } else if (arg == "--interactive") interactive = true; + else if (arg == "--login-screen") { + login_screen = true; + interactive = true; + } else if (arg == "--live-server" && i+1 < argc) { + live_server_spec = argv[++i]; + interactive = true; + } + else if (arg == "--width" && i+1 < argc) init_width = std::stoi(argv[++i]); + else if (arg == "--height" && i+1 < argc) init_height = std::stoi(argv[++i]); + else if (arg == "--draws" && i+1 < argc) { + draw_count = std::stoul(argv[++i]); + synthetic_specified = true; + } + else if (arg == "--triangles-per-draw" && i+1 < argc) { + triangles_per_draw = std::stoul(argv[++i]); + synthetic_specified = true; + } + else if (arg == "--capture" && i+1 < argc) { + capture_path = argv[++i]; + synthetic_specified = true; + } + else if (arg == "--capture-out" && i+1 < argc) capture_out = argv[++i]; + else if (arg == "--live-client" && i+1 < argc) live_client_dir = argv[++i]; + else if (arg == "--fake-mobs" && i+1 < argc) fake_mobs = std::stoi(argv[++i]); + else if (arg == "--animate-first-draw") { + animate_first_draw = true; + synthetic_specified = true; + } + else if (arg == "--animate-bones") { + animate_bones = true; + synthetic_specified = true; + } + else if (arg == "--no-vsync") vsync = false; + else if (arg == "--gpu-skinning") gpu_skinning = true; + else if (arg == "--no-gpu-skinning") gpu_skinning = false; + else if (arg == "--no-terrain") native_terrain = false; + else if (arg == "--mobile") mobile_mode = true; + else throw std::runtime_error( + "usage: mt_native_render [--frames N] [--interactive] [--login-screen] " + "[--live-server HOST:AUTH_PORT:GAME_PORT] [--width W] [--height H] " + "[--draws N] [--triangles-per-draw N] [--capture FILE] [--animate-first-draw] " + "[--animate-bones] [--no-vsync] [--live-client DIR] [--fake-mobs N] " + "[--gpu-skinning|--no-gpu-skinning] [--no-terrain] [--mobile] [--capture-out FILE]"); + } + if (live_client_dir.empty() && !synthetic_specified) { + const char* env_client = std::getenv("MT_40250_CLIENT"); + if (env_client && *env_client && access((std::string(env_client) + "/pack/Index").c_str(), R_OK) == 0) { + live_client_dir = env_client; + } else if (access("Client/pack/Index", R_OK) == 0) { + live_client_dir = "Client"; + } else if (access("../Client/pack/Index", R_OK) == 0) { + live_client_dir = "../Client"; + } else if (access("../../Client/pack/Index", R_OK) == 0) { + live_client_dir = "../../Client"; + } +#ifdef __ANDROID__ + if (live_client_dir.empty()) { + char* pref = SDL_GetPrefPath("mtgodot", "native-render"); + if (pref) { + const std::string candidate = std::string(pref) + "Client"; + if (access((candidate + "/pack/Index").c_str(), R_OK) == 0) + live_client_dir = candidate; + SDL_free(pref); + } + } +#endif + if (!live_client_dir.empty() && !frames_specified) { + interactive = true; + } + } + if (interactive && !frames_specified) frames = 0; + if (frames < 0 || (!interactive && frames == 0) || draw_count == 0 || draw_count > 4096 || + triangles_per_draw == 0 || triangles_per_draw > 10000) + throw std::runtime_error("invalid frame/draw/triangle count"); + VulkanWindow renderer(vsync, init_width, init_height); + if (mobile_mode) { + renderer.touch_controller.set_enabled(true); + renderer.touch_controller.update_screen_size(init_width, init_height); + } + if (!live_client_dir.empty()) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + return run_live_client( + renderer, + live_client_dir, + frames, + fake_mobs, + gpu_skinning, + native_terrain, + login_screen, + live_server_spec, + capture_out); +#else + throw std::runtime_error("mt_native_render was built without port_platform (--live-client unavailable)"); +#endif + } + native_draw_capture::Capture capture{}; + if (capture_path.empty()) { + capture.draws = make_test_draws(draw_count, triangles_per_draw); + } else { + capture = native_draw_capture::read_capture(capture_path); + } + auto& draws = capture.draws; + if (animate_first_draw && (draws.empty() || draws.front().positions.empty())) + throw std::runtime_error("first draw has no geometry to animate"); + const auto start = std::chrono::steady_clock::now(); + int completed = 0; + std::vector frame_ms; + if (frames > 0) frame_ms.reserve(static_cast(frames)); + while ((frames == 0 || completed < frames) && renderer.poll()) { + const auto frame_start = std::chrono::steady_clock::now(); + if (animate_first_draw && completed > 0) { + draws.front().positions[0] += 0.0001f; + ++draws.front().geometry_revision; + } + if (animate_bones && completed > 0) { + const float offset = std::sin(float(completed) * 0.1f) * 0.5f; + for (auto& d : draws) { + for (std::size_t b = 12; b < d.bone_matrices.size(); b += 16) + d.bone_matrices[b] += offset; + } + } + renderer.render(draws, capture.textures, capture.ui_width, capture.ui_height, capture.ui_commands); + frame_ms.push_back(std::chrono::duration( + std::chrono::steady_clock::now() - frame_start).count()); + ++completed; + } + renderer.finish_gpu_timings(); + const auto ms = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); + print_summary(renderer, completed, ms, 0.0, std::move(frame_ms)); + return (frames == 0 || completed == frames) ? 0 : 2; + } catch (const std::exception& error) { + std::cerr << "native renderer: " << error.what() << '\n'; + return 1; + } +} diff --git a/native_render/native.frag b/native_render/native.frag new file mode 100644 index 00000000..0c4db623 --- /dev/null +++ b/native_render/native.frag @@ -0,0 +1,139 @@ +#version 450 +layout(location = 0) in vec4 in_color; +layout(location = 1) in vec2 in_uv; +layout(location = 2) in vec2 in_mask_uv; +layout(location = 3) in float in_fog; +layout(location = 0) out vec4 out_color; + +layout(set = 0, binding = 0) uniform sampler2D tex_sampler; +layout(set = 0, binding = 1) uniform sampler2D mask_sampler; + +layout(push_constant) uniform DrawConstants { + mat4 mvp; + vec4 tint_color; + vec4 ambient_emissive; + vec4 light_dir; + vec4 light_diffuse; +} draw; + +layout(set = 1, binding = 1, std430) readonly buffer FixedFunctionState { + uvec4 stage0_color; + uvec4 stage0_alpha; + uvec4 stage1_color; + uvec4 stage1_alpha; + vec4 fog_color; + vec4 fog_params; + vec4 texture_factor; + mat4 world_view; + uvec4 flags; +} fixed_state; + +vec4 stage_arg(uint selector, vec4 diffuse, vec4 current, vec4 texel) { + uint source = selector & 15u; + vec4 value = source == 0u ? diffuse : + source == 1u ? current : + source == 2u ? texel : + source == 3u ? fixed_state.texture_factor : vec4(1.0); + if ((selector & 16u) != 0u) value = vec4(1.0) - value; + if ((selector & 32u) != 0u) value.rgb = vec3(value.a); + return value; +} + +vec4 apply_op(uint op, vec4 a, vec4 b, vec4 current, vec4 diffuse, vec4 texel) { + if (op == 2u) return a; // SELECTARG1 + if (op == 3u) return b; // SELECTARG2 + if (op == 4u) return a * b; // MODULATE + if (op == 5u) return 2.0 * a * b; // MODULATE2X + if (op == 6u) return 4.0 * a * b; // MODULATE4X + if (op == 7u) return a + b; // ADD + if (op == 8u) return a + b - vec4(0.5); // ADDSIGNED + if (op == 9u) return 2.0 * (a + b - vec4(0.5)); // ADDSIGNED2X + if (op == 10u) return a - b; // SUBTRACT + if (op == 11u) return a + b - a * b; // ADDSMOOTH + if (op == 12u) return mix(b, a, diffuse.a); + if (op == 13u) return mix(b, a, texel.a); + if (op == 14u) return mix(b, a, fixed_state.texture_factor.a); + if (op == 15u) return a + b * (1.0 - texel.a); + if (op == 16u) return mix(b, a, current.a); + if (op == 18u) return vec4(a.rgb + a.a * b.rgb, a.a); + if (op == 19u) return vec4(a.rgb * b.rgb + vec3(a.a), a.a); + if (op == 20u) return vec4((1.0 - a.a) * b.rgb + a.rgb, a.a); + if (op == 21u) return vec4((vec3(1.0) - a.rgb) * b.rgb + vec3(a.a), a.a); + return current; +} + +vec4 apply_stage(uvec4 color_state, uvec4 alpha_state, + vec4 diffuse, vec4 current, vec4 texel) { + if (color_state.x <= 1u) return current; + vec4 color_arg1 = stage_arg(color_state.y, diffuse, current, texel); + vec4 color_arg2 = stage_arg(color_state.z, diffuse, current, texel); + vec4 result = apply_op(color_state.x, color_arg1, color_arg2, current, diffuse, texel); + if (alpha_state.x > 1u) { + vec4 alpha_arg1 = stage_arg(alpha_state.y, diffuse, current, texel); + vec4 alpha_arg2 = stage_arg(alpha_state.z, diffuse, current, texel); + result.a = apply_op(alpha_state.x, alpha_arg1, alpha_arg2, current, diffuse, texel).a; + } else if (alpha_state.x == 1u) { + result.a = current.a; + } + return clamp(result, 0.0, 1.0); +} + +void main() { + vec4 tex0 = texture(tex_sampler, in_uv); + if (fixed_state.flags.x == 2u) { + vec3 shadow = tex0.rgb; + if (fixed_state.flags.z != 0u && fixed_state.stage1_color.x == 4u) + shadow *= texture(mask_sampler, in_mask_uv).rgb; + shadow = mix(fixed_state.fog_color.rgb, shadow, in_fog); + if (all(greaterThanEqual(shadow, vec3(0.997)))) discard; + out_color = vec4(shadow, 1.0); + return; + } + if (fixed_state.flags.x != 0u) { + vec4 diffuse = in_color; + vec4 color = apply_stage(fixed_state.stage0_color, fixed_state.stage0_alpha, + diffuse, diffuse, tex0); + if (fixed_state.flags.z != 0u && fixed_state.stage1_color.x > 1u) { + vec4 tex1 = texture(mask_sampler, in_mask_uv); + color = apply_stage(fixed_state.stage1_color, fixed_state.stage1_alpha, + diffuse, color, tex1); + } + if (fixed_state.stage1_alpha.w != 0u) { + float ref = draw.ambient_emissive.a; + uint func = fixed_state.stage0_alpha.w; + bool passes = func == 1u ? false : + func == 2u ? color.a < ref : + func == 3u ? abs(color.a - ref) < (0.5 / 255.0) : + func == 4u ? color.a <= ref : + func == 5u ? color.a > ref : + func == 6u ? abs(color.a - ref) >= (0.5 / 255.0) : + func == 7u ? color.a >= ref : true; + if (!passes) discard; + } + color.rgb = mix(fixed_state.fog_color.rgb, color.rgb, in_fog); + out_color = color; + return; + } + vec4 color = in_color * tex0; + if (draw.light_dir.w < -0.4) { + vec4 mask_val = texture(mask_sampler, in_mask_uv); + if (draw.light_dir.w < -0.8) { + if (in_mask_uv.x < 0.0 || in_mask_uv.x > 1.0 || in_mask_uv.y < 0.0 || in_mask_uv.y > 1.0) { + discard; + } + color.rgb *= mask_val.rgb; + color.a = mask_val.a * in_color.a; + } else { + color.a = mask_val.a * in_color.a; + if (color.a <= 0.003) discard; + } + } else if (draw.light_dir.w > 1.001) { + // Specular sphere-map stage 1 (D3DTOP_MODULATEALPHA_ADDCOLOR) + float spec_power = draw.light_dir.w - 1.0; + vec4 spec_map = texture(mask_sampler, in_mask_uv); + color.rgb = min(color.rgb + (tex0.a * spec_power) * spec_map.rgb, vec3(1.5)); + color.a = 1.0; + } + if (color.a <= draw.ambient_emissive.a) discard; + out_color = color; +} diff --git a/native_render/native.vert b/native_render/native.vert new file mode 100644 index 00000000..07aa8add --- /dev/null +++ b/native_render/native.vert @@ -0,0 +1,151 @@ +#version 450 +layout(location = 0) in vec3 in_position; +layout(location = 1) in vec3 in_normal; +layout(location = 2) in vec2 in_uv; +layout(location = 3) in vec4 in_color; +layout(location = 4) in uvec4 in_joints; +layout(location = 5) in vec4 in_weights; +layout(location = 6) in vec2 in_mask_uv; +layout(location = 7) in float in_rhw; + +layout(location = 0) out vec4 out_color; +layout(location = 1) out vec2 out_uv; +layout(location = 2) out vec2 out_mask_uv; +layout(location = 3) out float out_fog; + +layout(set = 1, binding = 0, std430) readonly buffer BonePalette { + mat4 bones[]; +} bone_palette; + +layout(push_constant) uniform DrawConstants { + mat4 mvp; + vec4 tint_color; + vec4 ambient_emissive; + vec4 light_dir; + vec4 light_diffuse; +} draw; + +layout(set = 1, binding = 1, std430) readonly buffer FixedFunctionState { + uvec4 stage0_color; + uvec4 stage0_alpha; + uvec4 stage1_color; + uvec4 stage1_alpha; + vec4 fog_color; + vec4 fog_params; + vec4 texture_factor; + mat4 world_view; + uvec4 flags; + mat4 world; + uvec4 lighting_flags; + vec4 material_ambient; + vec4 material_emissive; + vec4 global_ambient; + vec4 light_position_type[2]; + vec4 light_direction_range[2]; + vec4 light_diffuse[2]; + vec4 light_ambient[2]; + vec4 light_attenuation[2]; + vec4 light_spot[2]; +} fixed_state; + +void main() { + vec3 pos = in_position; + vec3 nrm = in_normal; + if (draw.light_diffuse.w > 0.5) { + uint bone_base = uint(draw.light_diffuse.w - 0.5); + vec3 skinned_pos = vec3(0.0); + vec3 skinned_nrm = vec3(0.0); + float total_w = 0.0; + for (int k = 0; k < 4; ++k) { + float w = in_weights[k]; + if (w > 0.0) { + mat4 B = bone_palette.bones[bone_base + in_joints[k]]; + skinned_pos += w * (B * vec4(pos, 1.0)).xyz; + skinned_nrm += w * (mat3(B) * nrm); + total_w += w; + } + } + if (total_w > 0.0) { + pos = skinned_pos; + nrm = skinned_nrm; + } + } + // D3D row-vector matrices are uploaded row-major. GLSL reads the bytes as their transpose. + gl_Position = draw.mvp * vec4(pos, 1.0); + if (draw.light_diffuse.w < -0.5) { + float clip_w = in_rhw > 0.0 ? 1.0 / in_rhw : 1.0; + gl_Position.xyz *= clip_w; + gl_Position.w = clip_w; + } + gl_Position.y = -gl_Position.y; + vec3 n = length(nrm) > 1e-4 ? normalize(nrm) : vec3(0.0, 0.0, 1.0); + out_color = in_color; + if (fixed_state.lighting_flags.x != 0u) { + vec3 world_pos = (fixed_state.world * vec4(pos, 1.0)).xyz; + vec3 world_normal = normalize(mat3(fixed_state.world) * n); + vec3 mat_diffuse = (fixed_state.lighting_flags.y != 0u && fixed_state.lighting_flags.w == 1u) + ? in_color.rgb : draw.tint_color.rgb; + vec3 mat_ambient = (fixed_state.lighting_flags.y != 0u && fixed_state.lighting_flags.z == 1u) + ? in_color.rgb : fixed_state.material_ambient.rgb; + vec3 ambient_sum = fixed_state.global_ambient.rgb; + vec3 diffuse_sum = vec3(0.0); + for (int i = 0; i < 2; ++i) { + int type = int(fixed_state.light_position_type[i].w + 0.5); + if (type == 0) continue; + vec3 L; + float strength = 1.0; + if (type == 3) { + L = normalize(-fixed_state.light_direction_range[i].xyz); + } else { + vec3 to_light = fixed_state.light_position_type[i].xyz - world_pos; + float distance_to_light = length(to_light); + if (distance_to_light > fixed_state.light_direction_range[i].w || distance_to_light < 0.0001) + continue; + L = to_light / distance_to_light; + vec4 attenuation = fixed_state.light_attenuation[i]; + float denominator = attenuation.x + attenuation.y * distance_to_light + + attenuation.z * distance_to_light * distance_to_light; + strength = denominator > 0.0001 ? min(1.0 / denominator, 1.0) : 1.0; + if (type == 2) { + vec3 spot_dir = normalize(fixed_state.light_direction_range[i].xyz); + float cosine = dot(-L, spot_dir); + float inner = cos(fixed_state.light_spot[i].x * 0.5); + float outer = cos(fixed_state.light_spot[i].y * 0.5); + float cone = clamp((cosine - outer) / max(inner - outer, 0.0001), 0.0, 1.0); + strength *= pow(cone, max(attenuation.w, 0.0001)); + } + } + ambient_sum += fixed_state.light_ambient[i].rgb * strength; + diffuse_sum += fixed_state.light_diffuse[i].rgb * max(dot(world_normal, L), 0.0) * strength; + } + vec3 lit = fixed_state.material_emissive.rgb + mat_ambient * ambient_sum + mat_diffuse * diffuse_sum; + float alpha = (fixed_state.lighting_flags.y != 0u && fixed_state.lighting_flags.w == 1u) + ? in_color.a : draw.tint_color.a; + out_color = vec4(clamp(lit, 0.0, 1.0), alpha); + } + out_uv = in_uv; + if (draw.light_dir.w > 1.001) { + vec3 view_normal = normalize(mat3(fixed_state.world_view) * n); + vec3 view_pos = (fixed_state.world_view * vec4(pos, 1.0)).xyz; + vec3 view_dir = normalize(-view_pos); + vec3 reflected = reflect(-view_dir, view_normal); + out_mask_uv = reflected.xy * vec2(0.5, -0.5) + vec2(0.5); + } else { + out_mask_uv = in_mask_uv; + } + out_fog = 1.0; + if (fixed_state.flags.x != 0u && fixed_state.flags.w != 0u && draw.light_diffuse.w >= -0.5) { + vec3 eye = (fixed_state.world_view * vec4(pos, 1.0)).xyz; + float distance_to_eye = fixed_state.fog_params.w > 0.5 ? length(eye) : abs(eye.z); + if (fixed_state.flags.w == 3u) { + float span = fixed_state.fog_params.y - fixed_state.fog_params.x; + if (span > 0.0001) + out_fog = clamp((fixed_state.fog_params.y - distance_to_eye) / span, 0.0, 1.0); + } else if (fixed_state.flags.w == 1u) { + out_fog = clamp(exp(-fixed_state.fog_params.z * distance_to_eye), 0.0, 1.0); + } else if (fixed_state.flags.w == 2u) { + float fog_distance = fixed_state.fog_params.z * distance_to_eye; + out_fog = clamp(exp(-fog_distance * fog_distance), 0.0, 1.0); + } + } +} diff --git a/native_render/stb_image_impl.cpp b/native_render/stb_image_impl.cpp new file mode 100644 index 00000000..62e1723d --- /dev/null +++ b/native_render/stb_image_impl.cpp @@ -0,0 +1,7 @@ +#define STBI_NO_STDIO +#define STBI_ONLY_JPEG +#define STBI_ONLY_PNG +#define STBI_ONLY_BMP +#define STBI_MAX_DIMENSIONS 4096 +#define STB_IMAGE_IMPLEMENTATION +#include "stb_image.h" diff --git a/native_render/third_party/stb_image.h b/native_render/third_party/stb_image.h new file mode 100644 index 00000000..9eedabed --- /dev/null +++ b/native_render/third_party/stb_image.h @@ -0,0 +1,7988 @@ +/* stb_image - v2.30 - public domain image loader - http://nothings.org/stb + no warranty implied; use at your own risk + + Do this: + #define STB_IMAGE_IMPLEMENTATION + before you include this file in *one* C or C++ file to create the implementation. + + // i.e. it should look like this: + #include ... + #include ... + #include ... + #define STB_IMAGE_IMPLEMENTATION + #include "stb_image.h" + + You can #define STBI_ASSERT(x) before the #include to avoid using assert.h. + And #define STBI_MALLOC, STBI_REALLOC, and STBI_FREE to avoid using malloc,realloc,free + + + QUICK NOTES: + Primarily of interest to game developers and other people who can + avoid problematic images and only need the trivial interface + + JPEG baseline & progressive (12 bpc/arithmetic not supported, same as stock IJG lib) + PNG 1/2/4/8/16-bit-per-channel + + TGA (not sure what subset, if a subset) + BMP non-1bpp, non-RLE + PSD (composited view only, no extra channels, 8/16 bit-per-channel) + + GIF (*comp always reports as 4-channel) + HDR (radiance rgbE format) + PIC (Softimage PIC) + PNM (PPM and PGM binary only) + + Animated GIF still needs a proper API, but here's one way to do it: + http://gist.github.com/urraka/685d9a6340b26b830d49 + + - decode from memory or through FILE (define STBI_NO_STDIO to remove code) + - decode from arbitrary I/O callbacks + - SIMD acceleration on x86/x64 (SSE2) and ARM (NEON) + + Full documentation under "DOCUMENTATION" below. + + +LICENSE + + See end of file for license information. + +RECENT REVISION HISTORY: + + 2.30 (2024-05-31) avoid erroneous gcc warning + 2.29 (2023-05-xx) optimizations + 2.28 (2023-01-29) many error fixes, security errors, just tons of stuff + 2.27 (2021-07-11) document stbi_info better, 16-bit PNM support, bug fixes + 2.26 (2020-07-13) many minor fixes + 2.25 (2020-02-02) fix warnings + 2.24 (2020-02-02) fix warnings; thread-local failure_reason and flip_vertically + 2.23 (2019-08-11) fix clang static analysis warning + 2.22 (2019-03-04) gif fixes, fix warnings + 2.21 (2019-02-25) fix typo in comment + 2.20 (2019-02-07) support utf8 filenames in Windows; fix warnings and platform ifdefs + 2.19 (2018-02-11) fix warning + 2.18 (2018-01-30) fix warnings + 2.17 (2018-01-29) bugfix, 1-bit BMP, 16-bitness query, fix warnings + 2.16 (2017-07-23) all functions have 16-bit variants; optimizations; bugfixes + 2.15 (2017-03-18) fix png-1,2,4; all Imagenet JPGs; no runtime SSE detection on GCC + 2.14 (2017-03-03) remove deprecated STBI_JPEG_OLD; fixes for Imagenet JPGs + 2.13 (2016-12-04) experimental 16-bit API, only for PNG so far; fixes + 2.12 (2016-04-02) fix typo in 2.11 PSD fix that caused crashes + 2.11 (2016-04-02) 16-bit PNGS; enable SSE2 in non-gcc x64 + RGB-format JPEG; remove white matting in PSD; + allocate large structures on the stack; + correct channel count for PNG & BMP + 2.10 (2016-01-22) avoid warning introduced in 2.09 + 2.09 (2016-01-16) 16-bit TGA; comments in PNM files; STBI_REALLOC_SIZED + + See end of file for full revision history. + + + ============================ Contributors ========================= + + Image formats Extensions, features + Sean Barrett (jpeg, png, bmp) Jetro Lauha (stbi_info) + Nicolas Schulz (hdr, psd) Martin "SpartanJ" Golini (stbi_info) + Jonathan Dummer (tga) James "moose2000" Brown (iPhone PNG) + Jean-Marc Lienher (gif) Ben "Disch" Wenger (io callbacks) + Tom Seddon (pic) Omar Cornut (1/2/4-bit PNG) + Thatcher Ulrich (psd) Nicolas Guillemot (vertical flip) + Ken Miller (pgm, ppm) Richard Mitton (16-bit PSD) + github:urraka (animated gif) Junggon Kim (PNM comments) + Christopher Forseth (animated gif) Daniel Gibson (16-bit TGA) + socks-the-fox (16-bit PNG) + Jeremy Sawicki (handle all ImageNet JPGs) + Optimizations & bugfixes Mikhail Morozov (1-bit BMP) + Fabian "ryg" Giesen Anael Seghezzi (is-16-bit query) + Arseny Kapoulkine Simon Breuss (16-bit PNM) + John-Mark Allen + Carmelo J Fdez-Aguera + + Bug & warning fixes + Marc LeBlanc David Woo Guillaume George Martins Mozeiko + Christpher Lloyd Jerry Jansson Joseph Thomson Blazej Dariusz Roszkowski + Phil Jordan Dave Moore Roy Eltham + Hayaki Saito Nathan Reed Won Chun + Luke Graham Johan Duparc Nick Verigakis the Horde3D community + Thomas Ruf Ronny Chevalier github:rlyeh + Janez Zemva John Bartholomew Michal Cichon github:romigrou + Jonathan Blow Ken Hamada Tero Hanninen github:svdijk + Eugene Golushkov Laurent Gomila Cort Stratton github:snagar + Aruelien Pocheville Sergio Gonzalez Thibault Reuille github:Zelex + Cass Everitt Ryamond Barbiero github:grim210 + Paul Du Bois Engin Manap Aldo Culquicondor github:sammyhw + Philipp Wiesemann Dale Weiler Oriol Ferrer Mesia github:phprus + Josh Tobin Neil Bickford Matthew Gregan github:poppolopoppo + Julian Raschke Gregory Mullen Christian Floisand github:darealshinji + Baldur Karlsson Kevin Schmidt JR Smith github:Michaelangel007 + Brad Weinberger Matvey Cherevko github:mosra + Luca Sas Alexander Veselov Zack Middleton [reserved] + Ryan C. Gordon [reserved] [reserved] + DO NOT ADD YOUR NAME HERE + + Jacko Dirks + + To add your name to the credits, pick a random blank space in the middle and fill it. + 80% of merge conflicts on stb PRs are due to people adding their name at the end + of the credits. +*/ + +#ifndef STBI_INCLUDE_STB_IMAGE_H +#define STBI_INCLUDE_STB_IMAGE_H + +// DOCUMENTATION +// +// Limitations: +// - no 12-bit-per-channel JPEG +// - no JPEGs with arithmetic coding +// - GIF always returns *comp=4 +// +// Basic usage (see HDR discussion below for HDR usage): +// int x,y,n; +// unsigned char *data = stbi_load(filename, &x, &y, &n, 0); +// // ... process data if not NULL ... +// // ... x = width, y = height, n = # 8-bit components per pixel ... +// // ... replace '0' with '1'..'4' to force that many components per pixel +// // ... but 'n' will always be the number that it would have been if you said 0 +// stbi_image_free(data); +// +// Standard parameters: +// int *x -- outputs image width in pixels +// int *y -- outputs image height in pixels +// int *channels_in_file -- outputs # of image components in image file +// int desired_channels -- if non-zero, # of image components requested in result +// +// The return value from an image loader is an 'unsigned char *' which points +// to the pixel data, or NULL on an allocation failure or if the image is +// corrupt or invalid. The pixel data consists of *y scanlines of *x pixels, +// with each pixel consisting of N interleaved 8-bit components; the first +// pixel pointed to is top-left-most in the image. There is no padding between +// image scanlines or between pixels, regardless of format. The number of +// components N is 'desired_channels' if desired_channels is non-zero, or +// *channels_in_file otherwise. If desired_channels is non-zero, +// *channels_in_file has the number of components that _would_ have been +// output otherwise. E.g. if you set desired_channels to 4, you will always +// get RGBA output, but you can check *channels_in_file to see if it's trivially +// opaque because e.g. there were only 3 channels in the source image. +// +// An output image with N components has the following components interleaved +// in this order in each pixel: +// +// N=#comp components +// 1 grey +// 2 grey, alpha +// 3 red, green, blue +// 4 red, green, blue, alpha +// +// If image loading fails for any reason, the return value will be NULL, +// and *x, *y, *channels_in_file will be unchanged. The function +// stbi_failure_reason() can be queried for an extremely brief, end-user +// unfriendly explanation of why the load failed. Define STBI_NO_FAILURE_STRINGS +// to avoid compiling these strings at all, and STBI_FAILURE_USERMSG to get slightly +// more user-friendly ones. +// +// Paletted PNG, BMP, GIF, and PIC images are automatically depalettized. +// +// To query the width, height and component count of an image without having to +// decode the full file, you can use the stbi_info family of functions: +// +// int x,y,n,ok; +// ok = stbi_info(filename, &x, &y, &n); +// // returns ok=1 and sets x, y, n if image is a supported format, +// // 0 otherwise. +// +// Note that stb_image pervasively uses ints in its public API for sizes, +// including sizes of memory buffers. This is now part of the API and thus +// hard to change without causing breakage. As a result, the various image +// loaders all have certain limits on image size; these differ somewhat +// by format but generally boil down to either just under 2GB or just under +// 1GB. When the decoded image would be larger than this, stb_image decoding +// will fail. +// +// Additionally, stb_image will reject image files that have any of their +// dimensions set to a larger value than the configurable STBI_MAX_DIMENSIONS, +// which defaults to 2**24 = 16777216 pixels. Due to the above memory limit, +// the only way to have an image with such dimensions load correctly +// is for it to have a rather extreme aspect ratio. Either way, the +// assumption here is that such larger images are likely to be malformed +// or malicious. If you do need to load an image with individual dimensions +// larger than that, and it still fits in the overall size limit, you can +// #define STBI_MAX_DIMENSIONS on your own to be something larger. +// +// =========================================================================== +// +// UNICODE: +// +// If compiling for Windows and you wish to use Unicode filenames, compile +// with +// #define STBI_WINDOWS_UTF8 +// and pass utf8-encoded filenames. Call stbi_convert_wchar_to_utf8 to convert +// Windows wchar_t filenames to utf8. +// +// =========================================================================== +// +// Philosophy +// +// stb libraries are designed with the following priorities: +// +// 1. easy to use +// 2. easy to maintain +// 3. good performance +// +// Sometimes I let "good performance" creep up in priority over "easy to maintain", +// and for best performance I may provide less-easy-to-use APIs that give higher +// performance, in addition to the easy-to-use ones. Nevertheless, it's important +// to keep in mind that from the standpoint of you, a client of this library, +// all you care about is #1 and #3, and stb libraries DO NOT emphasize #3 above all. +// +// Some secondary priorities arise directly from the first two, some of which +// provide more explicit reasons why performance can't be emphasized. +// +// - Portable ("ease of use") +// - Small source code footprint ("easy to maintain") +// - No dependencies ("ease of use") +// +// =========================================================================== +// +// I/O callbacks +// +// I/O callbacks allow you to read from arbitrary sources, like packaged +// files or some other source. Data read from callbacks are processed +// through a small internal buffer (currently 128 bytes) to try to reduce +// overhead. +// +// The three functions you must define are "read" (reads some bytes of data), +// "skip" (skips some bytes of data), "eof" (reports if the stream is at the end). +// +// =========================================================================== +// +// SIMD support +// +// The JPEG decoder will try to automatically use SIMD kernels on x86 when +// supported by the compiler. For ARM Neon support, you must explicitly +// request it. +// +// (The old do-it-yourself SIMD API is no longer supported in the current +// code.) +// +// On x86, SSE2 will automatically be used when available based on a run-time +// test; if not, the generic C versions are used as a fall-back. On ARM targets, +// the typical path is to have separate builds for NEON and non-NEON devices +// (at least this is true for iOS and Android). Therefore, the NEON support is +// toggled by a build flag: define STBI_NEON to get NEON loops. +// +// If for some reason you do not want to use any of SIMD code, or if +// you have issues compiling it, you can disable it entirely by +// defining STBI_NO_SIMD. +// +// =========================================================================== +// +// HDR image support (disable by defining STBI_NO_HDR) +// +// stb_image supports loading HDR images in general, and currently the Radiance +// .HDR file format specifically. You can still load any file through the existing +// interface; if you attempt to load an HDR file, it will be automatically remapped +// to LDR, assuming gamma 2.2 and an arbitrary scale factor defaulting to 1; +// both of these constants can be reconfigured through this interface: +// +// stbi_hdr_to_ldr_gamma(2.2f); +// stbi_hdr_to_ldr_scale(1.0f); +// +// (note, do not use _inverse_ constants; stbi_image will invert them +// appropriately). +// +// Additionally, there is a new, parallel interface for loading files as +// (linear) floats to preserve the full dynamic range: +// +// float *data = stbi_loadf(filename, &x, &y, &n, 0); +// +// If you load LDR images through this interface, those images will +// be promoted to floating point values, run through the inverse of +// constants corresponding to the above: +// +// stbi_ldr_to_hdr_scale(1.0f); +// stbi_ldr_to_hdr_gamma(2.2f); +// +// Finally, given a filename (or an open file or memory block--see header +// file for details) containing image data, you can query for the "most +// appropriate" interface to use (that is, whether the image is HDR or +// not), using: +// +// stbi_is_hdr(char *filename); +// +// =========================================================================== +// +// iPhone PNG support: +// +// We optionally support converting iPhone-formatted PNGs (which store +// premultiplied BGRA) back to RGB, even though they're internally encoded +// differently. To enable this conversion, call +// stbi_convert_iphone_png_to_rgb(1). +// +// Call stbi_set_unpremultiply_on_load(1) as well to force a divide per +// pixel to remove any premultiplied alpha *only* if the image file explicitly +// says there's premultiplied data (currently only happens in iPhone images, +// and only if iPhone convert-to-rgb processing is on). +// +// =========================================================================== +// +// ADDITIONAL CONFIGURATION +// +// - You can suppress implementation of any of the decoders to reduce +// your code footprint by #defining one or more of the following +// symbols before creating the implementation. +// +// STBI_NO_JPEG +// STBI_NO_PNG +// STBI_NO_BMP +// STBI_NO_PSD +// STBI_NO_TGA +// STBI_NO_GIF +// STBI_NO_HDR +// STBI_NO_PIC +// STBI_NO_PNM (.ppm and .pgm) +// +// - You can request *only* certain decoders and suppress all other ones +// (this will be more forward-compatible, as addition of new decoders +// doesn't require you to disable them explicitly): +// +// STBI_ONLY_JPEG +// STBI_ONLY_PNG +// STBI_ONLY_BMP +// STBI_ONLY_PSD +// STBI_ONLY_TGA +// STBI_ONLY_GIF +// STBI_ONLY_HDR +// STBI_ONLY_PIC +// STBI_ONLY_PNM (.ppm and .pgm) +// +// - If you use STBI_NO_PNG (or _ONLY_ without PNG), and you still +// want the zlib decoder to be available, #define STBI_SUPPORT_ZLIB +// +// - If you define STBI_MAX_DIMENSIONS, stb_image will reject images greater +// than that size (in either width or height) without further processing. +// This is to let programs in the wild set an upper bound to prevent +// denial-of-service attacks on untrusted data, as one could generate a +// valid image of gigantic dimensions and force stb_image to allocate a +// huge block of memory and spend disproportionate time decoding it. By +// default this is set to (1 << 24), which is 16777216, but that's still +// very big. + +#ifndef STBI_NO_STDIO +#include +#endif // STBI_NO_STDIO + +#define STBI_VERSION 1 + +enum +{ + STBI_default = 0, // only used for desired_channels + + STBI_grey = 1, + STBI_grey_alpha = 2, + STBI_rgb = 3, + STBI_rgb_alpha = 4 +}; + +#include +typedef unsigned char stbi_uc; +typedef unsigned short stbi_us; + +#ifdef __cplusplus +extern "C" { +#endif + +#ifndef STBIDEF +#ifdef STB_IMAGE_STATIC +#define STBIDEF static +#else +#define STBIDEF extern +#endif +#endif + +////////////////////////////////////////////////////////////////////////////// +// +// PRIMARY API - works on images of any type +// + +// +// load image by filename, open file, or memory buffer +// + +typedef struct +{ + int (*read) (void *user,char *data,int size); // fill 'data' with 'size' bytes. return number of bytes actually read + void (*skip) (void *user,int n); // skip the next 'n' bytes, or 'unget' the last -n bytes if negative + int (*eof) (void *user); // returns nonzero if we are at end of file/data +} stbi_io_callbacks; + +//////////////////////////////////// +// +// 8-bits-per-channel interface +// + +STBIDEF stbi_uc *stbi_load_from_memory (stbi_uc const *buffer, int len , int *x, int *y, int *channels_in_file, int desired_channels); +STBIDEF stbi_uc *stbi_load_from_callbacks(stbi_io_callbacks const *clbk , void *user, int *x, int *y, int *channels_in_file, int desired_channels); + +#ifndef STBI_NO_STDIO +STBIDEF stbi_uc *stbi_load (char const *filename, int *x, int *y, int *channels_in_file, int desired_channels); +STBIDEF stbi_uc *stbi_load_from_file (FILE *f, int *x, int *y, int *channels_in_file, int desired_channels); +// for stbi_load_from_file, file pointer is left pointing immediately after image +#endif + +#ifndef STBI_NO_GIF +STBIDEF stbi_uc *stbi_load_gif_from_memory(stbi_uc const *buffer, int len, int **delays, int *x, int *y, int *z, int *comp, int req_comp); +#endif + +#ifdef STBI_WINDOWS_UTF8 +STBIDEF int stbi_convert_wchar_to_utf8(char *buffer, size_t bufferlen, const wchar_t* input); +#endif + +//////////////////////////////////// +// +// 16-bits-per-channel interface +// + +STBIDEF stbi_us *stbi_load_16_from_memory (stbi_uc const *buffer, int len, int *x, int *y, int *channels_in_file, int desired_channels); +STBIDEF stbi_us *stbi_load_16_from_callbacks(stbi_io_callbacks const *clbk, void *user, int *x, int *y, int *channels_in_file, int desired_channels); + +#ifndef STBI_NO_STDIO +STBIDEF stbi_us *stbi_load_16 (char const *filename, int *x, int *y, int *channels_in_file, int desired_channels); +STBIDEF stbi_us *stbi_load_from_file_16(FILE *f, int *x, int *y, int *channels_in_file, int desired_channels); +#endif + +//////////////////////////////////// +// +// float-per-channel interface +// +#ifndef STBI_NO_LINEAR + STBIDEF float *stbi_loadf_from_memory (stbi_uc const *buffer, int len, int *x, int *y, int *channels_in_file, int desired_channels); + STBIDEF float *stbi_loadf_from_callbacks (stbi_io_callbacks const *clbk, void *user, int *x, int *y, int *channels_in_file, int desired_channels); + + #ifndef STBI_NO_STDIO + STBIDEF float *stbi_loadf (char const *filename, int *x, int *y, int *channels_in_file, int desired_channels); + STBIDEF float *stbi_loadf_from_file (FILE *f, int *x, int *y, int *channels_in_file, int desired_channels); + #endif +#endif + +#ifndef STBI_NO_HDR + STBIDEF void stbi_hdr_to_ldr_gamma(float gamma); + STBIDEF void stbi_hdr_to_ldr_scale(float scale); +#endif // STBI_NO_HDR + +#ifndef STBI_NO_LINEAR + STBIDEF void stbi_ldr_to_hdr_gamma(float gamma); + STBIDEF void stbi_ldr_to_hdr_scale(float scale); +#endif // STBI_NO_LINEAR + +// stbi_is_hdr is always defined, but always returns false if STBI_NO_HDR +STBIDEF int stbi_is_hdr_from_callbacks(stbi_io_callbacks const *clbk, void *user); +STBIDEF int stbi_is_hdr_from_memory(stbi_uc const *buffer, int len); +#ifndef STBI_NO_STDIO +STBIDEF int stbi_is_hdr (char const *filename); +STBIDEF int stbi_is_hdr_from_file(FILE *f); +#endif // STBI_NO_STDIO + + +// get a VERY brief reason for failure +// on most compilers (and ALL modern mainstream compilers) this is threadsafe +STBIDEF const char *stbi_failure_reason (void); + +// free the loaded image -- this is just free() +STBIDEF void stbi_image_free (void *retval_from_stbi_load); + +// get image dimensions & components without fully decoding +STBIDEF int stbi_info_from_memory(stbi_uc const *buffer, int len, int *x, int *y, int *comp); +STBIDEF int stbi_info_from_callbacks(stbi_io_callbacks const *clbk, void *user, int *x, int *y, int *comp); +STBIDEF int stbi_is_16_bit_from_memory(stbi_uc const *buffer, int len); +STBIDEF int stbi_is_16_bit_from_callbacks(stbi_io_callbacks const *clbk, void *user); + +#ifndef STBI_NO_STDIO +STBIDEF int stbi_info (char const *filename, int *x, int *y, int *comp); +STBIDEF int stbi_info_from_file (FILE *f, int *x, int *y, int *comp); +STBIDEF int stbi_is_16_bit (char const *filename); +STBIDEF int stbi_is_16_bit_from_file(FILE *f); +#endif + + + +// for image formats that explicitly notate that they have premultiplied alpha, +// we just return the colors as stored in the file. set this flag to force +// unpremultiplication. results are undefined if the unpremultiply overflow. +STBIDEF void stbi_set_unpremultiply_on_load(int flag_true_if_should_unpremultiply); + +// indicate whether we should process iphone images back to canonical format, +// or just pass them through "as-is" +STBIDEF void stbi_convert_iphone_png_to_rgb(int flag_true_if_should_convert); + +// flip the image vertically, so the first pixel in the output array is the bottom left +STBIDEF void stbi_set_flip_vertically_on_load(int flag_true_if_should_flip); + +// as above, but only applies to images loaded on the thread that calls the function +// this function is only available if your compiler supports thread-local variables; +// calling it will fail to link if your compiler doesn't +STBIDEF void stbi_set_unpremultiply_on_load_thread(int flag_true_if_should_unpremultiply); +STBIDEF void stbi_convert_iphone_png_to_rgb_thread(int flag_true_if_should_convert); +STBIDEF void stbi_set_flip_vertically_on_load_thread(int flag_true_if_should_flip); + +// ZLIB client - used by PNG, available for other purposes + +STBIDEF char *stbi_zlib_decode_malloc_guesssize(const char *buffer, int len, int initial_size, int *outlen); +STBIDEF char *stbi_zlib_decode_malloc_guesssize_headerflag(const char *buffer, int len, int initial_size, int *outlen, int parse_header); +STBIDEF char *stbi_zlib_decode_malloc(const char *buffer, int len, int *outlen); +STBIDEF int stbi_zlib_decode_buffer(char *obuffer, int olen, const char *ibuffer, int ilen); + +STBIDEF char *stbi_zlib_decode_noheader_malloc(const char *buffer, int len, int *outlen); +STBIDEF int stbi_zlib_decode_noheader_buffer(char *obuffer, int olen, const char *ibuffer, int ilen); + + +#ifdef __cplusplus +} +#endif + +// +// +//// end header file ///////////////////////////////////////////////////// +#endif // STBI_INCLUDE_STB_IMAGE_H + +#ifdef STB_IMAGE_IMPLEMENTATION + +#if defined(STBI_ONLY_JPEG) || defined(STBI_ONLY_PNG) || defined(STBI_ONLY_BMP) \ + || defined(STBI_ONLY_TGA) || defined(STBI_ONLY_GIF) || defined(STBI_ONLY_PSD) \ + || defined(STBI_ONLY_HDR) || defined(STBI_ONLY_PIC) || defined(STBI_ONLY_PNM) \ + || defined(STBI_ONLY_ZLIB) + #ifndef STBI_ONLY_JPEG + #define STBI_NO_JPEG + #endif + #ifndef STBI_ONLY_PNG + #define STBI_NO_PNG + #endif + #ifndef STBI_ONLY_BMP + #define STBI_NO_BMP + #endif + #ifndef STBI_ONLY_PSD + #define STBI_NO_PSD + #endif + #ifndef STBI_ONLY_TGA + #define STBI_NO_TGA + #endif + #ifndef STBI_ONLY_GIF + #define STBI_NO_GIF + #endif + #ifndef STBI_ONLY_HDR + #define STBI_NO_HDR + #endif + #ifndef STBI_ONLY_PIC + #define STBI_NO_PIC + #endif + #ifndef STBI_ONLY_PNM + #define STBI_NO_PNM + #endif +#endif + +#if defined(STBI_NO_PNG) && !defined(STBI_SUPPORT_ZLIB) && !defined(STBI_NO_ZLIB) +#define STBI_NO_ZLIB +#endif + + +#include +#include // ptrdiff_t on osx +#include +#include +#include + +#if !defined(STBI_NO_LINEAR) || !defined(STBI_NO_HDR) +#include // ldexp, pow +#endif + +#ifndef STBI_NO_STDIO +#include +#endif + +#ifndef STBI_ASSERT +#include +#define STBI_ASSERT(x) assert(x) +#endif + +#ifdef __cplusplus +#define STBI_EXTERN extern "C" +#else +#define STBI_EXTERN extern +#endif + + +#ifndef _MSC_VER + #ifdef __cplusplus + #define stbi_inline inline + #else + #define stbi_inline + #endif +#else + #define stbi_inline __forceinline +#endif + +#ifndef STBI_NO_THREAD_LOCALS + #if defined(__cplusplus) && __cplusplus >= 201103L + #define STBI_THREAD_LOCAL thread_local + #elif defined(__GNUC__) && __GNUC__ < 5 + #define STBI_THREAD_LOCAL __thread + #elif defined(_MSC_VER) + #define STBI_THREAD_LOCAL __declspec(thread) + #elif defined (__STDC_VERSION__) && __STDC_VERSION__ >= 201112L && !defined(__STDC_NO_THREADS__) + #define STBI_THREAD_LOCAL _Thread_local + #endif + + #ifndef STBI_THREAD_LOCAL + #if defined(__GNUC__) + #define STBI_THREAD_LOCAL __thread + #endif + #endif +#endif + +#if defined(_MSC_VER) || defined(__SYMBIAN32__) +typedef unsigned short stbi__uint16; +typedef signed short stbi__int16; +typedef unsigned int stbi__uint32; +typedef signed int stbi__int32; +#else +#include +typedef uint16_t stbi__uint16; +typedef int16_t stbi__int16; +typedef uint32_t stbi__uint32; +typedef int32_t stbi__int32; +#endif + +// should produce compiler error if size is wrong +typedef unsigned char validate_uint32[sizeof(stbi__uint32)==4 ? 1 : -1]; + +#ifdef _MSC_VER +#define STBI_NOTUSED(v) (void)(v) +#else +#define STBI_NOTUSED(v) (void)sizeof(v) +#endif + +#ifdef _MSC_VER +#define STBI_HAS_LROTL +#endif + +#ifdef STBI_HAS_LROTL + #define stbi_lrot(x,y) _lrotl(x,y) +#else + #define stbi_lrot(x,y) (((x) << (y)) | ((x) >> (-(y) & 31))) +#endif + +#if defined(STBI_MALLOC) && defined(STBI_FREE) && (defined(STBI_REALLOC) || defined(STBI_REALLOC_SIZED)) +// ok +#elif !defined(STBI_MALLOC) && !defined(STBI_FREE) && !defined(STBI_REALLOC) && !defined(STBI_REALLOC_SIZED) +// ok +#else +#error "Must define all or none of STBI_MALLOC, STBI_FREE, and STBI_REALLOC (or STBI_REALLOC_SIZED)." +#endif + +#ifndef STBI_MALLOC +#define STBI_MALLOC(sz) malloc(sz) +#define STBI_REALLOC(p,newsz) realloc(p,newsz) +#define STBI_FREE(p) free(p) +#endif + +#ifndef STBI_REALLOC_SIZED +#define STBI_REALLOC_SIZED(p,oldsz,newsz) STBI_REALLOC(p,newsz) +#endif + +// x86/x64 detection +#if defined(__x86_64__) || defined(_M_X64) +#define STBI__X64_TARGET +#elif defined(__i386) || defined(_M_IX86) +#define STBI__X86_TARGET +#endif + +#if defined(__GNUC__) && defined(STBI__X86_TARGET) && !defined(__SSE2__) && !defined(STBI_NO_SIMD) +// gcc doesn't support sse2 intrinsics unless you compile with -msse2, +// which in turn means it gets to use SSE2 everywhere. This is unfortunate, +// but previous attempts to provide the SSE2 functions with runtime +// detection caused numerous issues. The way architecture extensions are +// exposed in GCC/Clang is, sadly, not really suited for one-file libs. +// New behavior: if compiled with -msse2, we use SSE2 without any +// detection; if not, we don't use it at all. +#define STBI_NO_SIMD +#endif + +#if defined(__MINGW32__) && defined(STBI__X86_TARGET) && !defined(STBI_MINGW_ENABLE_SSE2) && !defined(STBI_NO_SIMD) +// Note that __MINGW32__ doesn't actually mean 32-bit, so we have to avoid STBI__X64_TARGET +// +// 32-bit MinGW wants ESP to be 16-byte aligned, but this is not in the +// Windows ABI and VC++ as well as Windows DLLs don't maintain that invariant. +// As a result, enabling SSE2 on 32-bit MinGW is dangerous when not +// simultaneously enabling "-mstackrealign". +// +// See https://github.com/nothings/stb/issues/81 for more information. +// +// So default to no SSE2 on 32-bit MinGW. If you've read this far and added +// -mstackrealign to your build settings, feel free to #define STBI_MINGW_ENABLE_SSE2. +#define STBI_NO_SIMD +#endif + +#if !defined(STBI_NO_SIMD) && (defined(STBI__X86_TARGET) || defined(STBI__X64_TARGET)) +#define STBI_SSE2 +#include + +#ifdef _MSC_VER + +#if _MSC_VER >= 1400 // not VC6 +#include // __cpuid +static int stbi__cpuid3(void) +{ + int info[4]; + __cpuid(info,1); + return info[3]; +} +#else +static int stbi__cpuid3(void) +{ + int res; + __asm { + mov eax,1 + cpuid + mov res,edx + } + return res; +} +#endif + +#define STBI_SIMD_ALIGN(type, name) __declspec(align(16)) type name + +#if !defined(STBI_NO_JPEG) && defined(STBI_SSE2) +static int stbi__sse2_available(void) +{ + int info3 = stbi__cpuid3(); + return ((info3 >> 26) & 1) != 0; +} +#endif + +#else // assume GCC-style if not VC++ +#define STBI_SIMD_ALIGN(type, name) type name __attribute__((aligned(16))) + +#if !defined(STBI_NO_JPEG) && defined(STBI_SSE2) +static int stbi__sse2_available(void) +{ + // If we're even attempting to compile this on GCC/Clang, that means + // -msse2 is on, which means the compiler is allowed to use SSE2 + // instructions at will, and so are we. + return 1; +} +#endif + +#endif +#endif + +// ARM NEON +#if defined(STBI_NO_SIMD) && defined(STBI_NEON) +#undef STBI_NEON +#endif + +#ifdef STBI_NEON +#include +#ifdef _MSC_VER +#define STBI_SIMD_ALIGN(type, name) __declspec(align(16)) type name +#else +#define STBI_SIMD_ALIGN(type, name) type name __attribute__((aligned(16))) +#endif +#endif + +#ifndef STBI_SIMD_ALIGN +#define STBI_SIMD_ALIGN(type, name) type name +#endif + +#ifndef STBI_MAX_DIMENSIONS +#define STBI_MAX_DIMENSIONS (1 << 24) +#endif + +/////////////////////////////////////////////// +// +// stbi__context struct and start_xxx functions + +// stbi__context structure is our basic context used by all images, so it +// contains all the IO context, plus some basic image information +typedef struct +{ + stbi__uint32 img_x, img_y; + int img_n, img_out_n; + + stbi_io_callbacks io; + void *io_user_data; + + int read_from_callbacks; + int buflen; + stbi_uc buffer_start[128]; + int callback_already_read; + + stbi_uc *img_buffer, *img_buffer_end; + stbi_uc *img_buffer_original, *img_buffer_original_end; +} stbi__context; + + +static void stbi__refill_buffer(stbi__context *s); + +// initialize a memory-decode context +static void stbi__start_mem(stbi__context *s, stbi_uc const *buffer, int len) +{ + s->io.read = NULL; + s->read_from_callbacks = 0; + s->callback_already_read = 0; + s->img_buffer = s->img_buffer_original = (stbi_uc *) buffer; + s->img_buffer_end = s->img_buffer_original_end = (stbi_uc *) buffer+len; +} + +// initialize a callback-based context +static void stbi__start_callbacks(stbi__context *s, stbi_io_callbacks *c, void *user) +{ + s->io = *c; + s->io_user_data = user; + s->buflen = sizeof(s->buffer_start); + s->read_from_callbacks = 1; + s->callback_already_read = 0; + s->img_buffer = s->img_buffer_original = s->buffer_start; + stbi__refill_buffer(s); + s->img_buffer_original_end = s->img_buffer_end; +} + +#ifndef STBI_NO_STDIO + +static int stbi__stdio_read(void *user, char *data, int size) +{ + return (int) fread(data,1,size,(FILE*) user); +} + +static void stbi__stdio_skip(void *user, int n) +{ + int ch; + fseek((FILE*) user, n, SEEK_CUR); + ch = fgetc((FILE*) user); /* have to read a byte to reset feof()'s flag */ + if (ch != EOF) { + ungetc(ch, (FILE *) user); /* push byte back onto stream if valid. */ + } +} + +static int stbi__stdio_eof(void *user) +{ + return feof((FILE*) user) || ferror((FILE *) user); +} + +static stbi_io_callbacks stbi__stdio_callbacks = +{ + stbi__stdio_read, + stbi__stdio_skip, + stbi__stdio_eof, +}; + +static void stbi__start_file(stbi__context *s, FILE *f) +{ + stbi__start_callbacks(s, &stbi__stdio_callbacks, (void *) f); +} + +//static void stop_file(stbi__context *s) { } + +#endif // !STBI_NO_STDIO + +static void stbi__rewind(stbi__context *s) +{ + // conceptually rewind SHOULD rewind to the beginning of the stream, + // but we just rewind to the beginning of the initial buffer, because + // we only use it after doing 'test', which only ever looks at at most 92 bytes + s->img_buffer = s->img_buffer_original; + s->img_buffer_end = s->img_buffer_original_end; +} + +enum +{ + STBI_ORDER_RGB, + STBI_ORDER_BGR +}; + +typedef struct +{ + int bits_per_channel; + int num_channels; + int channel_order; +} stbi__result_info; + +#ifndef STBI_NO_JPEG +static int stbi__jpeg_test(stbi__context *s); +static void *stbi__jpeg_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__jpeg_info(stbi__context *s, int *x, int *y, int *comp); +#endif + +#ifndef STBI_NO_PNG +static int stbi__png_test(stbi__context *s); +static void *stbi__png_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__png_info(stbi__context *s, int *x, int *y, int *comp); +static int stbi__png_is16(stbi__context *s); +#endif + +#ifndef STBI_NO_BMP +static int stbi__bmp_test(stbi__context *s); +static void *stbi__bmp_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__bmp_info(stbi__context *s, int *x, int *y, int *comp); +#endif + +#ifndef STBI_NO_TGA +static int stbi__tga_test(stbi__context *s); +static void *stbi__tga_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__tga_info(stbi__context *s, int *x, int *y, int *comp); +#endif + +#ifndef STBI_NO_PSD +static int stbi__psd_test(stbi__context *s); +static void *stbi__psd_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri, int bpc); +static int stbi__psd_info(stbi__context *s, int *x, int *y, int *comp); +static int stbi__psd_is16(stbi__context *s); +#endif + +#ifndef STBI_NO_HDR +static int stbi__hdr_test(stbi__context *s); +static float *stbi__hdr_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__hdr_info(stbi__context *s, int *x, int *y, int *comp); +#endif + +#ifndef STBI_NO_PIC +static int stbi__pic_test(stbi__context *s); +static void *stbi__pic_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__pic_info(stbi__context *s, int *x, int *y, int *comp); +#endif + +#ifndef STBI_NO_GIF +static int stbi__gif_test(stbi__context *s); +static void *stbi__gif_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static void *stbi__load_gif_main(stbi__context *s, int **delays, int *x, int *y, int *z, int *comp, int req_comp); +static int stbi__gif_info(stbi__context *s, int *x, int *y, int *comp); +#endif + +#ifndef STBI_NO_PNM +static int stbi__pnm_test(stbi__context *s); +static void *stbi__pnm_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri); +static int stbi__pnm_info(stbi__context *s, int *x, int *y, int *comp); +static int stbi__pnm_is16(stbi__context *s); +#endif + +static +#ifdef STBI_THREAD_LOCAL +STBI_THREAD_LOCAL +#endif +const char *stbi__g_failure_reason; + +STBIDEF const char *stbi_failure_reason(void) +{ + return stbi__g_failure_reason; +} + +#ifndef STBI_NO_FAILURE_STRINGS +static int stbi__err(const char *str) +{ + stbi__g_failure_reason = str; + return 0; +} +#endif + +static void *stbi__malloc(size_t size) +{ + return STBI_MALLOC(size); +} + +// stb_image uses ints pervasively, including for offset calculations. +// therefore the largest decoded image size we can support with the +// current code, even on 64-bit targets, is INT_MAX. this is not a +// significant limitation for the intended use case. +// +// we do, however, need to make sure our size calculations don't +// overflow. hence a few helper functions for size calculations that +// multiply integers together, making sure that they're non-negative +// and no overflow occurs. + +// return 1 if the sum is valid, 0 on overflow. +// negative terms are considered invalid. +static int stbi__addsizes_valid(int a, int b) +{ + if (b < 0) return 0; + // now 0 <= b <= INT_MAX, hence also + // 0 <= INT_MAX - b <= INTMAX. + // And "a + b <= INT_MAX" (which might overflow) is the + // same as a <= INT_MAX - b (no overflow) + return a <= INT_MAX - b; +} + +// returns 1 if the product is valid, 0 on overflow. +// negative factors are considered invalid. +static int stbi__mul2sizes_valid(int a, int b) +{ + if (a < 0 || b < 0) return 0; + if (b == 0) return 1; // mul-by-0 is always safe + // portable way to check for no overflows in a*b + return a <= INT_MAX/b; +} + +#if !defined(STBI_NO_JPEG) || !defined(STBI_NO_PNG) || !defined(STBI_NO_TGA) || !defined(STBI_NO_HDR) +// returns 1 if "a*b + add" has no negative terms/factors and doesn't overflow +static int stbi__mad2sizes_valid(int a, int b, int add) +{ + return stbi__mul2sizes_valid(a, b) && stbi__addsizes_valid(a*b, add); +} +#endif + +// returns 1 if "a*b*c + add" has no negative terms/factors and doesn't overflow +static int stbi__mad3sizes_valid(int a, int b, int c, int add) +{ + return stbi__mul2sizes_valid(a, b) && stbi__mul2sizes_valid(a*b, c) && + stbi__addsizes_valid(a*b*c, add); +} + +// returns 1 if "a*b*c*d + add" has no negative terms/factors and doesn't overflow +#if !defined(STBI_NO_LINEAR) || !defined(STBI_NO_HDR) || !defined(STBI_NO_PNM) +static int stbi__mad4sizes_valid(int a, int b, int c, int d, int add) +{ + return stbi__mul2sizes_valid(a, b) && stbi__mul2sizes_valid(a*b, c) && + stbi__mul2sizes_valid(a*b*c, d) && stbi__addsizes_valid(a*b*c*d, add); +} +#endif + +#if !defined(STBI_NO_JPEG) || !defined(STBI_NO_PNG) || !defined(STBI_NO_TGA) || !defined(STBI_NO_HDR) +// mallocs with size overflow checking +static void *stbi__malloc_mad2(int a, int b, int add) +{ + if (!stbi__mad2sizes_valid(a, b, add)) return NULL; + return stbi__malloc(a*b + add); +} +#endif + +static void *stbi__malloc_mad3(int a, int b, int c, int add) +{ + if (!stbi__mad3sizes_valid(a, b, c, add)) return NULL; + return stbi__malloc(a*b*c + add); +} + +#if !defined(STBI_NO_LINEAR) || !defined(STBI_NO_HDR) || !defined(STBI_NO_PNM) +static void *stbi__malloc_mad4(int a, int b, int c, int d, int add) +{ + if (!stbi__mad4sizes_valid(a, b, c, d, add)) return NULL; + return stbi__malloc(a*b*c*d + add); +} +#endif + +// returns 1 if the sum of two signed ints is valid (between -2^31 and 2^31-1 inclusive), 0 on overflow. +static int stbi__addints_valid(int a, int b) +{ + if ((a >= 0) != (b >= 0)) return 1; // a and b have different signs, so no overflow + if (a < 0 && b < 0) return a >= INT_MIN - b; // same as a + b >= INT_MIN; INT_MIN - b cannot overflow since b < 0. + return a <= INT_MAX - b; +} + +// returns 1 if the product of two ints fits in a signed short, 0 on overflow. +static int stbi__mul2shorts_valid(int a, int b) +{ + if (b == 0 || b == -1) return 1; // multiplication by 0 is always 0; check for -1 so SHRT_MIN/b doesn't overflow + if ((a >= 0) == (b >= 0)) return a <= SHRT_MAX/b; // product is positive, so similar to mul2sizes_valid + if (b < 0) return a <= SHRT_MIN / b; // same as a * b >= SHRT_MIN + return a >= SHRT_MIN / b; +} + +// stbi__err - error +// stbi__errpf - error returning pointer to float +// stbi__errpuc - error returning pointer to unsigned char + +#ifdef STBI_NO_FAILURE_STRINGS + #define stbi__err(x,y) 0 +#elif defined(STBI_FAILURE_USERMSG) + #define stbi__err(x,y) stbi__err(y) +#else + #define stbi__err(x,y) stbi__err(x) +#endif + +#define stbi__errpf(x,y) ((float *)(size_t) (stbi__err(x,y)?NULL:NULL)) +#define stbi__errpuc(x,y) ((unsigned char *)(size_t) (stbi__err(x,y)?NULL:NULL)) + +STBIDEF void stbi_image_free(void *retval_from_stbi_load) +{ + STBI_FREE(retval_from_stbi_load); +} + +#ifndef STBI_NO_LINEAR +static float *stbi__ldr_to_hdr(stbi_uc *data, int x, int y, int comp); +#endif + +#ifndef STBI_NO_HDR +static stbi_uc *stbi__hdr_to_ldr(float *data, int x, int y, int comp); +#endif + +static int stbi__vertically_flip_on_load_global = 0; + +STBIDEF void stbi_set_flip_vertically_on_load(int flag_true_if_should_flip) +{ + stbi__vertically_flip_on_load_global = flag_true_if_should_flip; +} + +#ifndef STBI_THREAD_LOCAL +#define stbi__vertically_flip_on_load stbi__vertically_flip_on_load_global +#else +static STBI_THREAD_LOCAL int stbi__vertically_flip_on_load_local, stbi__vertically_flip_on_load_set; + +STBIDEF void stbi_set_flip_vertically_on_load_thread(int flag_true_if_should_flip) +{ + stbi__vertically_flip_on_load_local = flag_true_if_should_flip; + stbi__vertically_flip_on_load_set = 1; +} + +#define stbi__vertically_flip_on_load (stbi__vertically_flip_on_load_set \ + ? stbi__vertically_flip_on_load_local \ + : stbi__vertically_flip_on_load_global) +#endif // STBI_THREAD_LOCAL + +static void *stbi__load_main(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri, int bpc) +{ + memset(ri, 0, sizeof(*ri)); // make sure it's initialized if we add new fields + ri->bits_per_channel = 8; // default is 8 so most paths don't have to be changed + ri->channel_order = STBI_ORDER_RGB; // all current input & output are this, but this is here so we can add BGR order + ri->num_channels = 0; + + // test the formats with a very explicit header first (at least a FOURCC + // or distinctive magic number first) + #ifndef STBI_NO_PNG + if (stbi__png_test(s)) return stbi__png_load(s,x,y,comp,req_comp, ri); + #endif + #ifndef STBI_NO_BMP + if (stbi__bmp_test(s)) return stbi__bmp_load(s,x,y,comp,req_comp, ri); + #endif + #ifndef STBI_NO_GIF + if (stbi__gif_test(s)) return stbi__gif_load(s,x,y,comp,req_comp, ri); + #endif + #ifndef STBI_NO_PSD + if (stbi__psd_test(s)) return stbi__psd_load(s,x,y,comp,req_comp, ri, bpc); + #else + STBI_NOTUSED(bpc); + #endif + #ifndef STBI_NO_PIC + if (stbi__pic_test(s)) return stbi__pic_load(s,x,y,comp,req_comp, ri); + #endif + + // then the formats that can end up attempting to load with just 1 or 2 + // bytes matching expectations; these are prone to false positives, so + // try them later + #ifndef STBI_NO_JPEG + if (stbi__jpeg_test(s)) return stbi__jpeg_load(s,x,y,comp,req_comp, ri); + #endif + #ifndef STBI_NO_PNM + if (stbi__pnm_test(s)) return stbi__pnm_load(s,x,y,comp,req_comp, ri); + #endif + + #ifndef STBI_NO_HDR + if (stbi__hdr_test(s)) { + float *hdr = stbi__hdr_load(s, x,y,comp,req_comp, ri); + return stbi__hdr_to_ldr(hdr, *x, *y, req_comp ? req_comp : *comp); + } + #endif + + #ifndef STBI_NO_TGA + // test tga last because it's a crappy test! + if (stbi__tga_test(s)) + return stbi__tga_load(s,x,y,comp,req_comp, ri); + #endif + + return stbi__errpuc("unknown image type", "Image not of any known type, or corrupt"); +} + +static stbi_uc *stbi__convert_16_to_8(stbi__uint16 *orig, int w, int h, int channels) +{ + int i; + int img_len = w * h * channels; + stbi_uc *reduced; + + reduced = (stbi_uc *) stbi__malloc(img_len); + if (reduced == NULL) return stbi__errpuc("outofmem", "Out of memory"); + + for (i = 0; i < img_len; ++i) + reduced[i] = (stbi_uc)((orig[i] >> 8) & 0xFF); // top half of each byte is sufficient approx of 16->8 bit scaling + + STBI_FREE(orig); + return reduced; +} + +static stbi__uint16 *stbi__convert_8_to_16(stbi_uc *orig, int w, int h, int channels) +{ + int i; + int img_len = w * h * channels; + stbi__uint16 *enlarged; + + enlarged = (stbi__uint16 *) stbi__malloc(img_len*2); + if (enlarged == NULL) return (stbi__uint16 *) stbi__errpuc("outofmem", "Out of memory"); + + for (i = 0; i < img_len; ++i) + enlarged[i] = (stbi__uint16)((orig[i] << 8) + orig[i]); // replicate to high and low byte, maps 0->0, 255->0xffff + + STBI_FREE(orig); + return enlarged; +} + +static void stbi__vertical_flip(void *image, int w, int h, int bytes_per_pixel) +{ + int row; + size_t bytes_per_row = (size_t)w * bytes_per_pixel; + stbi_uc temp[2048]; + stbi_uc *bytes = (stbi_uc *)image; + + for (row = 0; row < (h>>1); row++) { + stbi_uc *row0 = bytes + row*bytes_per_row; + stbi_uc *row1 = bytes + (h - row - 1)*bytes_per_row; + // swap row0 with row1 + size_t bytes_left = bytes_per_row; + while (bytes_left) { + size_t bytes_copy = (bytes_left < sizeof(temp)) ? bytes_left : sizeof(temp); + memcpy(temp, row0, bytes_copy); + memcpy(row0, row1, bytes_copy); + memcpy(row1, temp, bytes_copy); + row0 += bytes_copy; + row1 += bytes_copy; + bytes_left -= bytes_copy; + } + } +} + +#ifndef STBI_NO_GIF +static void stbi__vertical_flip_slices(void *image, int w, int h, int z, int bytes_per_pixel) +{ + int slice; + int slice_size = w * h * bytes_per_pixel; + + stbi_uc *bytes = (stbi_uc *)image; + for (slice = 0; slice < z; ++slice) { + stbi__vertical_flip(bytes, w, h, bytes_per_pixel); + bytes += slice_size; + } +} +#endif + +static unsigned char *stbi__load_and_postprocess_8bit(stbi__context *s, int *x, int *y, int *comp, int req_comp) +{ + stbi__result_info ri; + void *result = stbi__load_main(s, x, y, comp, req_comp, &ri, 8); + + if (result == NULL) + return NULL; + + // it is the responsibility of the loaders to make sure we get either 8 or 16 bit. + STBI_ASSERT(ri.bits_per_channel == 8 || ri.bits_per_channel == 16); + + if (ri.bits_per_channel != 8) { + result = stbi__convert_16_to_8((stbi__uint16 *) result, *x, *y, req_comp == 0 ? *comp : req_comp); + ri.bits_per_channel = 8; + } + + // @TODO: move stbi__convert_format to here + + if (stbi__vertically_flip_on_load) { + int channels = req_comp ? req_comp : *comp; + stbi__vertical_flip(result, *x, *y, channels * sizeof(stbi_uc)); + } + + return (unsigned char *) result; +} + +static stbi__uint16 *stbi__load_and_postprocess_16bit(stbi__context *s, int *x, int *y, int *comp, int req_comp) +{ + stbi__result_info ri; + void *result = stbi__load_main(s, x, y, comp, req_comp, &ri, 16); + + if (result == NULL) + return NULL; + + // it is the responsibility of the loaders to make sure we get either 8 or 16 bit. + STBI_ASSERT(ri.bits_per_channel == 8 || ri.bits_per_channel == 16); + + if (ri.bits_per_channel != 16) { + result = stbi__convert_8_to_16((stbi_uc *) result, *x, *y, req_comp == 0 ? *comp : req_comp); + ri.bits_per_channel = 16; + } + + // @TODO: move stbi__convert_format16 to here + // @TODO: special case RGB-to-Y (and RGBA-to-YA) for 8-bit-to-16-bit case to keep more precision + + if (stbi__vertically_flip_on_load) { + int channels = req_comp ? req_comp : *comp; + stbi__vertical_flip(result, *x, *y, channels * sizeof(stbi__uint16)); + } + + return (stbi__uint16 *) result; +} + +#if !defined(STBI_NO_HDR) && !defined(STBI_NO_LINEAR) +static void stbi__float_postprocess(float *result, int *x, int *y, int *comp, int req_comp) +{ + if (stbi__vertically_flip_on_load && result != NULL) { + int channels = req_comp ? req_comp : *comp; + stbi__vertical_flip(result, *x, *y, channels * sizeof(float)); + } +} +#endif + +#ifndef STBI_NO_STDIO + +#if defined(_WIN32) && defined(STBI_WINDOWS_UTF8) +STBI_EXTERN __declspec(dllimport) int __stdcall MultiByteToWideChar(unsigned int cp, unsigned long flags, const char *str, int cbmb, wchar_t *widestr, int cchwide); +STBI_EXTERN __declspec(dllimport) int __stdcall WideCharToMultiByte(unsigned int cp, unsigned long flags, const wchar_t *widestr, int cchwide, char *str, int cbmb, const char *defchar, int *used_default); +#endif + +#if defined(_WIN32) && defined(STBI_WINDOWS_UTF8) +STBIDEF int stbi_convert_wchar_to_utf8(char *buffer, size_t bufferlen, const wchar_t* input) +{ + return WideCharToMultiByte(65001 /* UTF8 */, 0, input, -1, buffer, (int) bufferlen, NULL, NULL); +} +#endif + +static FILE *stbi__fopen(char const *filename, char const *mode) +{ + FILE *f; +#if defined(_WIN32) && defined(STBI_WINDOWS_UTF8) + wchar_t wMode[64]; + wchar_t wFilename[1024]; + if (0 == MultiByteToWideChar(65001 /* UTF8 */, 0, filename, -1, wFilename, sizeof(wFilename)/sizeof(*wFilename))) + return 0; + + if (0 == MultiByteToWideChar(65001 /* UTF8 */, 0, mode, -1, wMode, sizeof(wMode)/sizeof(*wMode))) + return 0; + +#if defined(_MSC_VER) && _MSC_VER >= 1400 + if (0 != _wfopen_s(&f, wFilename, wMode)) + f = 0; +#else + f = _wfopen(wFilename, wMode); +#endif + +#elif defined(_MSC_VER) && _MSC_VER >= 1400 + if (0 != fopen_s(&f, filename, mode)) + f=0; +#else + f = fopen(filename, mode); +#endif + return f; +} + + +STBIDEF stbi_uc *stbi_load(char const *filename, int *x, int *y, int *comp, int req_comp) +{ + FILE *f = stbi__fopen(filename, "rb"); + unsigned char *result; + if (!f) return stbi__errpuc("can't fopen", "Unable to open file"); + result = stbi_load_from_file(f,x,y,comp,req_comp); + fclose(f); + return result; +} + +STBIDEF stbi_uc *stbi_load_from_file(FILE *f, int *x, int *y, int *comp, int req_comp) +{ + unsigned char *result; + stbi__context s; + stbi__start_file(&s,f); + result = stbi__load_and_postprocess_8bit(&s,x,y,comp,req_comp); + if (result) { + // need to 'unget' all the characters in the IO buffer + fseek(f, - (int) (s.img_buffer_end - s.img_buffer), SEEK_CUR); + } + return result; +} + +STBIDEF stbi__uint16 *stbi_load_from_file_16(FILE *f, int *x, int *y, int *comp, int req_comp) +{ + stbi__uint16 *result; + stbi__context s; + stbi__start_file(&s,f); + result = stbi__load_and_postprocess_16bit(&s,x,y,comp,req_comp); + if (result) { + // need to 'unget' all the characters in the IO buffer + fseek(f, - (int) (s.img_buffer_end - s.img_buffer), SEEK_CUR); + } + return result; +} + +STBIDEF stbi_us *stbi_load_16(char const *filename, int *x, int *y, int *comp, int req_comp) +{ + FILE *f = stbi__fopen(filename, "rb"); + stbi__uint16 *result; + if (!f) return (stbi_us *) stbi__errpuc("can't fopen", "Unable to open file"); + result = stbi_load_from_file_16(f,x,y,comp,req_comp); + fclose(f); + return result; +} + + +#endif //!STBI_NO_STDIO + +STBIDEF stbi_us *stbi_load_16_from_memory(stbi_uc const *buffer, int len, int *x, int *y, int *channels_in_file, int desired_channels) +{ + stbi__context s; + stbi__start_mem(&s,buffer,len); + return stbi__load_and_postprocess_16bit(&s,x,y,channels_in_file,desired_channels); +} + +STBIDEF stbi_us *stbi_load_16_from_callbacks(stbi_io_callbacks const *clbk, void *user, int *x, int *y, int *channels_in_file, int desired_channels) +{ + stbi__context s; + stbi__start_callbacks(&s, (stbi_io_callbacks *)clbk, user); + return stbi__load_and_postprocess_16bit(&s,x,y,channels_in_file,desired_channels); +} + +STBIDEF stbi_uc *stbi_load_from_memory(stbi_uc const *buffer, int len, int *x, int *y, int *comp, int req_comp) +{ + stbi__context s; + stbi__start_mem(&s,buffer,len); + return stbi__load_and_postprocess_8bit(&s,x,y,comp,req_comp); +} + +STBIDEF stbi_uc *stbi_load_from_callbacks(stbi_io_callbacks const *clbk, void *user, int *x, int *y, int *comp, int req_comp) +{ + stbi__context s; + stbi__start_callbacks(&s, (stbi_io_callbacks *) clbk, user); + return stbi__load_and_postprocess_8bit(&s,x,y,comp,req_comp); +} + +#ifndef STBI_NO_GIF +STBIDEF stbi_uc *stbi_load_gif_from_memory(stbi_uc const *buffer, int len, int **delays, int *x, int *y, int *z, int *comp, int req_comp) +{ + unsigned char *result; + stbi__context s; + stbi__start_mem(&s,buffer,len); + + result = (unsigned char*) stbi__load_gif_main(&s, delays, x, y, z, comp, req_comp); + if (stbi__vertically_flip_on_load) { + stbi__vertical_flip_slices( result, *x, *y, *z, *comp ); + } + + return result; +} +#endif + +#ifndef STBI_NO_LINEAR +static float *stbi__loadf_main(stbi__context *s, int *x, int *y, int *comp, int req_comp) +{ + unsigned char *data; + #ifndef STBI_NO_HDR + if (stbi__hdr_test(s)) { + stbi__result_info ri; + float *hdr_data = stbi__hdr_load(s,x,y,comp,req_comp, &ri); + if (hdr_data) + stbi__float_postprocess(hdr_data,x,y,comp,req_comp); + return hdr_data; + } + #endif + data = stbi__load_and_postprocess_8bit(s, x, y, comp, req_comp); + if (data) + return stbi__ldr_to_hdr(data, *x, *y, req_comp ? req_comp : *comp); + return stbi__errpf("unknown image type", "Image not of any known type, or corrupt"); +} + +STBIDEF float *stbi_loadf_from_memory(stbi_uc const *buffer, int len, int *x, int *y, int *comp, int req_comp) +{ + stbi__context s; + stbi__start_mem(&s,buffer,len); + return stbi__loadf_main(&s,x,y,comp,req_comp); +} + +STBIDEF float *stbi_loadf_from_callbacks(stbi_io_callbacks const *clbk, void *user, int *x, int *y, int *comp, int req_comp) +{ + stbi__context s; + stbi__start_callbacks(&s, (stbi_io_callbacks *) clbk, user); + return stbi__loadf_main(&s,x,y,comp,req_comp); +} + +#ifndef STBI_NO_STDIO +STBIDEF float *stbi_loadf(char const *filename, int *x, int *y, int *comp, int req_comp) +{ + float *result; + FILE *f = stbi__fopen(filename, "rb"); + if (!f) return stbi__errpf("can't fopen", "Unable to open file"); + result = stbi_loadf_from_file(f,x,y,comp,req_comp); + fclose(f); + return result; +} + +STBIDEF float *stbi_loadf_from_file(FILE *f, int *x, int *y, int *comp, int req_comp) +{ + stbi__context s; + stbi__start_file(&s,f); + return stbi__loadf_main(&s,x,y,comp,req_comp); +} +#endif // !STBI_NO_STDIO + +#endif // !STBI_NO_LINEAR + +// these is-hdr-or-not is defined independent of whether STBI_NO_LINEAR is +// defined, for API simplicity; if STBI_NO_LINEAR is defined, it always +// reports false! + +STBIDEF int stbi_is_hdr_from_memory(stbi_uc const *buffer, int len) +{ + #ifndef STBI_NO_HDR + stbi__context s; + stbi__start_mem(&s,buffer,len); + return stbi__hdr_test(&s); + #else + STBI_NOTUSED(buffer); + STBI_NOTUSED(len); + return 0; + #endif +} + +#ifndef STBI_NO_STDIO +STBIDEF int stbi_is_hdr (char const *filename) +{ + FILE *f = stbi__fopen(filename, "rb"); + int result=0; + if (f) { + result = stbi_is_hdr_from_file(f); + fclose(f); + } + return result; +} + +STBIDEF int stbi_is_hdr_from_file(FILE *f) +{ + #ifndef STBI_NO_HDR + long pos = ftell(f); + int res; + stbi__context s; + stbi__start_file(&s,f); + res = stbi__hdr_test(&s); + fseek(f, pos, SEEK_SET); + return res; + #else + STBI_NOTUSED(f); + return 0; + #endif +} +#endif // !STBI_NO_STDIO + +STBIDEF int stbi_is_hdr_from_callbacks(stbi_io_callbacks const *clbk, void *user) +{ + #ifndef STBI_NO_HDR + stbi__context s; + stbi__start_callbacks(&s, (stbi_io_callbacks *) clbk, user); + return stbi__hdr_test(&s); + #else + STBI_NOTUSED(clbk); + STBI_NOTUSED(user); + return 0; + #endif +} + +#ifndef STBI_NO_LINEAR +static float stbi__l2h_gamma=2.2f, stbi__l2h_scale=1.0f; + +STBIDEF void stbi_ldr_to_hdr_gamma(float gamma) { stbi__l2h_gamma = gamma; } +STBIDEF void stbi_ldr_to_hdr_scale(float scale) { stbi__l2h_scale = scale; } +#endif + +static float stbi__h2l_gamma_i=1.0f/2.2f, stbi__h2l_scale_i=1.0f; + +STBIDEF void stbi_hdr_to_ldr_gamma(float gamma) { stbi__h2l_gamma_i = 1/gamma; } +STBIDEF void stbi_hdr_to_ldr_scale(float scale) { stbi__h2l_scale_i = 1/scale; } + + +////////////////////////////////////////////////////////////////////////////// +// +// Common code used by all image loaders +// + +enum +{ + STBI__SCAN_load=0, + STBI__SCAN_type, + STBI__SCAN_header +}; + +static void stbi__refill_buffer(stbi__context *s) +{ + int n = (s->io.read)(s->io_user_data,(char*)s->buffer_start,s->buflen); + s->callback_already_read += (int) (s->img_buffer - s->img_buffer_original); + if (n == 0) { + // at end of file, treat same as if from memory, but need to handle case + // where s->img_buffer isn't pointing to safe memory, e.g. 0-byte file + s->read_from_callbacks = 0; + s->img_buffer = s->buffer_start; + s->img_buffer_end = s->buffer_start+1; + *s->img_buffer = 0; + } else { + s->img_buffer = s->buffer_start; + s->img_buffer_end = s->buffer_start + n; + } +} + +stbi_inline static stbi_uc stbi__get8(stbi__context *s) +{ + if (s->img_buffer < s->img_buffer_end) + return *s->img_buffer++; + if (s->read_from_callbacks) { + stbi__refill_buffer(s); + return *s->img_buffer++; + } + return 0; +} + +#if defined(STBI_NO_JPEG) && defined(STBI_NO_HDR) && defined(STBI_NO_PIC) && defined(STBI_NO_PNM) +// nothing +#else +stbi_inline static int stbi__at_eof(stbi__context *s) +{ + if (s->io.read) { + if (!(s->io.eof)(s->io_user_data)) return 0; + // if feof() is true, check if buffer = end + // special case: we've only got the special 0 character at the end + if (s->read_from_callbacks == 0) return 1; + } + + return s->img_buffer >= s->img_buffer_end; +} +#endif + +#if defined(STBI_NO_JPEG) && defined(STBI_NO_PNG) && defined(STBI_NO_BMP) && defined(STBI_NO_PSD) && defined(STBI_NO_TGA) && defined(STBI_NO_GIF) && defined(STBI_NO_PIC) +// nothing +#else +static void stbi__skip(stbi__context *s, int n) +{ + if (n == 0) return; // already there! + if (n < 0) { + s->img_buffer = s->img_buffer_end; + return; + } + if (s->io.read) { + int blen = (int) (s->img_buffer_end - s->img_buffer); + if (blen < n) { + s->img_buffer = s->img_buffer_end; + (s->io.skip)(s->io_user_data, n - blen); + return; + } + } + s->img_buffer += n; +} +#endif + +#if defined(STBI_NO_PNG) && defined(STBI_NO_TGA) && defined(STBI_NO_HDR) && defined(STBI_NO_PNM) +// nothing +#else +static int stbi__getn(stbi__context *s, stbi_uc *buffer, int n) +{ + if (s->io.read) { + int blen = (int) (s->img_buffer_end - s->img_buffer); + if (blen < n) { + int res, count; + + memcpy(buffer, s->img_buffer, blen); + + count = (s->io.read)(s->io_user_data, (char*) buffer + blen, n - blen); + res = (count == (n-blen)); + s->img_buffer = s->img_buffer_end; + return res; + } + } + + if (s->img_buffer+n <= s->img_buffer_end) { + memcpy(buffer, s->img_buffer, n); + s->img_buffer += n; + return 1; + } else + return 0; +} +#endif + +#if defined(STBI_NO_JPEG) && defined(STBI_NO_PNG) && defined(STBI_NO_PSD) && defined(STBI_NO_PIC) +// nothing +#else +static int stbi__get16be(stbi__context *s) +{ + int z = stbi__get8(s); + return (z << 8) + stbi__get8(s); +} +#endif + +#if defined(STBI_NO_PNG) && defined(STBI_NO_PSD) && defined(STBI_NO_PIC) +// nothing +#else +static stbi__uint32 stbi__get32be(stbi__context *s) +{ + stbi__uint32 z = stbi__get16be(s); + return (z << 16) + stbi__get16be(s); +} +#endif + +#if defined(STBI_NO_BMP) && defined(STBI_NO_TGA) && defined(STBI_NO_GIF) +// nothing +#else +static int stbi__get16le(stbi__context *s) +{ + int z = stbi__get8(s); + return z + (stbi__get8(s) << 8); +} +#endif + +#ifndef STBI_NO_BMP +static stbi__uint32 stbi__get32le(stbi__context *s) +{ + stbi__uint32 z = stbi__get16le(s); + z += (stbi__uint32)stbi__get16le(s) << 16; + return z; +} +#endif + +#define STBI__BYTECAST(x) ((stbi_uc) ((x) & 255)) // truncate int to byte without warnings + +#if defined(STBI_NO_JPEG) && defined(STBI_NO_PNG) && defined(STBI_NO_BMP) && defined(STBI_NO_PSD) && defined(STBI_NO_TGA) && defined(STBI_NO_GIF) && defined(STBI_NO_PIC) && defined(STBI_NO_PNM) +// nothing +#else +////////////////////////////////////////////////////////////////////////////// +// +// generic converter from built-in img_n to req_comp +// individual types do this automatically as much as possible (e.g. jpeg +// does all cases internally since it needs to colorspace convert anyway, +// and it never has alpha, so very few cases ). png can automatically +// interleave an alpha=255 channel, but falls back to this for other cases +// +// assume data buffer is malloced, so malloc a new one and free that one +// only failure mode is malloc failing + +static stbi_uc stbi__compute_y(int r, int g, int b) +{ + return (stbi_uc) (((r*77) + (g*150) + (29*b)) >> 8); +} +#endif + +#if defined(STBI_NO_PNG) && defined(STBI_NO_BMP) && defined(STBI_NO_PSD) && defined(STBI_NO_TGA) && defined(STBI_NO_GIF) && defined(STBI_NO_PIC) && defined(STBI_NO_PNM) +// nothing +#else +static unsigned char *stbi__convert_format(unsigned char *data, int img_n, int req_comp, unsigned int x, unsigned int y) +{ + int i,j; + unsigned char *good; + + if (req_comp == img_n) return data; + STBI_ASSERT(req_comp >= 1 && req_comp <= 4); + + good = (unsigned char *) stbi__malloc_mad3(req_comp, x, y, 0); + if (good == NULL) { + STBI_FREE(data); + return stbi__errpuc("outofmem", "Out of memory"); + } + + for (j=0; j < (int) y; ++j) { + unsigned char *src = data + j * x * img_n ; + unsigned char *dest = good + j * x * req_comp; + + #define STBI__COMBO(a,b) ((a)*8+(b)) + #define STBI__CASE(a,b) case STBI__COMBO(a,b): for(i=x-1; i >= 0; --i, src += a, dest += b) + // convert source image with img_n components to one with req_comp components; + // avoid switch per pixel, so use switch per scanline and massive macros + switch (STBI__COMBO(img_n, req_comp)) { + STBI__CASE(1,2) { dest[0]=src[0]; dest[1]=255; } break; + STBI__CASE(1,3) { dest[0]=dest[1]=dest[2]=src[0]; } break; + STBI__CASE(1,4) { dest[0]=dest[1]=dest[2]=src[0]; dest[3]=255; } break; + STBI__CASE(2,1) { dest[0]=src[0]; } break; + STBI__CASE(2,3) { dest[0]=dest[1]=dest[2]=src[0]; } break; + STBI__CASE(2,4) { dest[0]=dest[1]=dest[2]=src[0]; dest[3]=src[1]; } break; + STBI__CASE(3,4) { dest[0]=src[0];dest[1]=src[1];dest[2]=src[2];dest[3]=255; } break; + STBI__CASE(3,1) { dest[0]=stbi__compute_y(src[0],src[1],src[2]); } break; + STBI__CASE(3,2) { dest[0]=stbi__compute_y(src[0],src[1],src[2]); dest[1] = 255; } break; + STBI__CASE(4,1) { dest[0]=stbi__compute_y(src[0],src[1],src[2]); } break; + STBI__CASE(4,2) { dest[0]=stbi__compute_y(src[0],src[1],src[2]); dest[1] = src[3]; } break; + STBI__CASE(4,3) { dest[0]=src[0];dest[1]=src[1];dest[2]=src[2]; } break; + default: STBI_ASSERT(0); STBI_FREE(data); STBI_FREE(good); return stbi__errpuc("unsupported", "Unsupported format conversion"); + } + #undef STBI__CASE + } + + STBI_FREE(data); + return good; +} +#endif + +#if defined(STBI_NO_PNG) && defined(STBI_NO_PSD) +// nothing +#else +static stbi__uint16 stbi__compute_y_16(int r, int g, int b) +{ + return (stbi__uint16) (((r*77) + (g*150) + (29*b)) >> 8); +} +#endif + +#if defined(STBI_NO_PNG) && defined(STBI_NO_PSD) +// nothing +#else +static stbi__uint16 *stbi__convert_format16(stbi__uint16 *data, int img_n, int req_comp, unsigned int x, unsigned int y) +{ + int i,j; + stbi__uint16 *good; + + if (req_comp == img_n) return data; + STBI_ASSERT(req_comp >= 1 && req_comp <= 4); + + good = (stbi__uint16 *) stbi__malloc(req_comp * x * y * 2); + if (good == NULL) { + STBI_FREE(data); + return (stbi__uint16 *) stbi__errpuc("outofmem", "Out of memory"); + } + + for (j=0; j < (int) y; ++j) { + stbi__uint16 *src = data + j * x * img_n ; + stbi__uint16 *dest = good + j * x * req_comp; + + #define STBI__COMBO(a,b) ((a)*8+(b)) + #define STBI__CASE(a,b) case STBI__COMBO(a,b): for(i=x-1; i >= 0; --i, src += a, dest += b) + // convert source image with img_n components to one with req_comp components; + // avoid switch per pixel, so use switch per scanline and massive macros + switch (STBI__COMBO(img_n, req_comp)) { + STBI__CASE(1,2) { dest[0]=src[0]; dest[1]=0xffff; } break; + STBI__CASE(1,3) { dest[0]=dest[1]=dest[2]=src[0]; } break; + STBI__CASE(1,4) { dest[0]=dest[1]=dest[2]=src[0]; dest[3]=0xffff; } break; + STBI__CASE(2,1) { dest[0]=src[0]; } break; + STBI__CASE(2,3) { dest[0]=dest[1]=dest[2]=src[0]; } break; + STBI__CASE(2,4) { dest[0]=dest[1]=dest[2]=src[0]; dest[3]=src[1]; } break; + STBI__CASE(3,4) { dest[0]=src[0];dest[1]=src[1];dest[2]=src[2];dest[3]=0xffff; } break; + STBI__CASE(3,1) { dest[0]=stbi__compute_y_16(src[0],src[1],src[2]); } break; + STBI__CASE(3,2) { dest[0]=stbi__compute_y_16(src[0],src[1],src[2]); dest[1] = 0xffff; } break; + STBI__CASE(4,1) { dest[0]=stbi__compute_y_16(src[0],src[1],src[2]); } break; + STBI__CASE(4,2) { dest[0]=stbi__compute_y_16(src[0],src[1],src[2]); dest[1] = src[3]; } break; + STBI__CASE(4,3) { dest[0]=src[0];dest[1]=src[1];dest[2]=src[2]; } break; + default: STBI_ASSERT(0); STBI_FREE(data); STBI_FREE(good); return (stbi__uint16*) stbi__errpuc("unsupported", "Unsupported format conversion"); + } + #undef STBI__CASE + } + + STBI_FREE(data); + return good; +} +#endif + +#ifndef STBI_NO_LINEAR +static float *stbi__ldr_to_hdr(stbi_uc *data, int x, int y, int comp) +{ + int i,k,n; + float *output; + if (!data) return NULL; + output = (float *) stbi__malloc_mad4(x, y, comp, sizeof(float), 0); + if (output == NULL) { STBI_FREE(data); return stbi__errpf("outofmem", "Out of memory"); } + // compute number of non-alpha components + if (comp & 1) n = comp; else n = comp-1; + for (i=0; i < x*y; ++i) { + for (k=0; k < n; ++k) { + output[i*comp + k] = (float) (pow(data[i*comp+k]/255.0f, stbi__l2h_gamma) * stbi__l2h_scale); + } + } + if (n < comp) { + for (i=0; i < x*y; ++i) { + output[i*comp + n] = data[i*comp + n]/255.0f; + } + } + STBI_FREE(data); + return output; +} +#endif + +#ifndef STBI_NO_HDR +#define stbi__float2int(x) ((int) (x)) +static stbi_uc *stbi__hdr_to_ldr(float *data, int x, int y, int comp) +{ + int i,k,n; + stbi_uc *output; + if (!data) return NULL; + output = (stbi_uc *) stbi__malloc_mad3(x, y, comp, 0); + if (output == NULL) { STBI_FREE(data); return stbi__errpuc("outofmem", "Out of memory"); } + // compute number of non-alpha components + if (comp & 1) n = comp; else n = comp-1; + for (i=0; i < x*y; ++i) { + for (k=0; k < n; ++k) { + float z = (float) pow(data[i*comp+k]*stbi__h2l_scale_i, stbi__h2l_gamma_i) * 255 + 0.5f; + if (z < 0) z = 0; + if (z > 255) z = 255; + output[i*comp + k] = (stbi_uc) stbi__float2int(z); + } + if (k < comp) { + float z = data[i*comp+k] * 255 + 0.5f; + if (z < 0) z = 0; + if (z > 255) z = 255; + output[i*comp + k] = (stbi_uc) stbi__float2int(z); + } + } + STBI_FREE(data); + return output; +} +#endif + +////////////////////////////////////////////////////////////////////////////// +// +// "baseline" JPEG/JFIF decoder +// +// simple implementation +// - doesn't support delayed output of y-dimension +// - simple interface (only one output format: 8-bit interleaved RGB) +// - doesn't try to recover corrupt jpegs +// - doesn't allow partial loading, loading multiple at once +// - still fast on x86 (copying globals into locals doesn't help x86) +// - allocates lots of intermediate memory (full size of all components) +// - non-interleaved case requires this anyway +// - allows good upsampling (see next) +// high-quality +// - upsampled channels are bilinearly interpolated, even across blocks +// - quality integer IDCT derived from IJG's 'slow' +// performance +// - fast huffman; reasonable integer IDCT +// - some SIMD kernels for common paths on targets with SSE2/NEON +// - uses a lot of intermediate memory, could cache poorly + +#ifndef STBI_NO_JPEG + +// huffman decoding acceleration +#define FAST_BITS 9 // larger handles more cases; smaller stomps less cache + +typedef struct +{ + stbi_uc fast[1 << FAST_BITS]; + // weirdly, repacking this into AoS is a 10% speed loss, instead of a win + stbi__uint16 code[256]; + stbi_uc values[256]; + stbi_uc size[257]; + unsigned int maxcode[18]; + int delta[17]; // old 'firstsymbol' - old 'firstcode' +} stbi__huffman; + +typedef struct +{ + stbi__context *s; + stbi__huffman huff_dc[4]; + stbi__huffman huff_ac[4]; + stbi__uint16 dequant[4][64]; + stbi__int16 fast_ac[4][1 << FAST_BITS]; + +// sizes for components, interleaved MCUs + int img_h_max, img_v_max; + int img_mcu_x, img_mcu_y; + int img_mcu_w, img_mcu_h; + +// definition of jpeg image component + struct + { + int id; + int h,v; + int tq; + int hd,ha; + int dc_pred; + + int x,y,w2,h2; + stbi_uc *data; + void *raw_data, *raw_coeff; + stbi_uc *linebuf; + short *coeff; // progressive only + int coeff_w, coeff_h; // number of 8x8 coefficient blocks + } img_comp[4]; + + stbi__uint32 code_buffer; // jpeg entropy-coded buffer + int code_bits; // number of valid bits + unsigned char marker; // marker seen while filling entropy buffer + int nomore; // flag if we saw a marker so must stop + + int progressive; + int spec_start; + int spec_end; + int succ_high; + int succ_low; + int eob_run; + int jfif; + int app14_color_transform; // Adobe APP14 tag + int rgb; + + int scan_n, order[4]; + int restart_interval, todo; + +// kernels + void (*idct_block_kernel)(stbi_uc *out, int out_stride, short data[64]); + void (*YCbCr_to_RGB_kernel)(stbi_uc *out, const stbi_uc *y, const stbi_uc *pcb, const stbi_uc *pcr, int count, int step); + stbi_uc *(*resample_row_hv_2_kernel)(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs); +} stbi__jpeg; + +static int stbi__build_huffman(stbi__huffman *h, int *count) +{ + int i,j,k=0; + unsigned int code; + // build size list for each symbol (from JPEG spec) + for (i=0; i < 16; ++i) { + for (j=0; j < count[i]; ++j) { + h->size[k++] = (stbi_uc) (i+1); + if(k >= 257) return stbi__err("bad size list","Corrupt JPEG"); + } + } + h->size[k] = 0; + + // compute actual symbols (from jpeg spec) + code = 0; + k = 0; + for(j=1; j <= 16; ++j) { + // compute delta to add to code to compute symbol id + h->delta[j] = k - code; + if (h->size[k] == j) { + while (h->size[k] == j) + h->code[k++] = (stbi__uint16) (code++); + if (code-1 >= (1u << j)) return stbi__err("bad code lengths","Corrupt JPEG"); + } + // compute largest code + 1 for this size, preshifted as needed later + h->maxcode[j] = code << (16-j); + code <<= 1; + } + h->maxcode[j] = 0xffffffff; + + // build non-spec acceleration table; 255 is flag for not-accelerated + memset(h->fast, 255, 1 << FAST_BITS); + for (i=0; i < k; ++i) { + int s = h->size[i]; + if (s <= FAST_BITS) { + int c = h->code[i] << (FAST_BITS-s); + int m = 1 << (FAST_BITS-s); + for (j=0; j < m; ++j) { + h->fast[c+j] = (stbi_uc) i; + } + } + } + return 1; +} + +// build a table that decodes both magnitude and value of small ACs in +// one go. +static void stbi__build_fast_ac(stbi__int16 *fast_ac, stbi__huffman *h) +{ + int i; + for (i=0; i < (1 << FAST_BITS); ++i) { + stbi_uc fast = h->fast[i]; + fast_ac[i] = 0; + if (fast < 255) { + int rs = h->values[fast]; + int run = (rs >> 4) & 15; + int magbits = rs & 15; + int len = h->size[fast]; + + if (magbits && len + magbits <= FAST_BITS) { + // magnitude code followed by receive_extend code + int k = ((i << len) & ((1 << FAST_BITS) - 1)) >> (FAST_BITS - magbits); + int m = 1 << (magbits - 1); + if (k < m) k += (~0U << magbits) + 1; + // if the result is small enough, we can fit it in fast_ac table + if (k >= -128 && k <= 127) + fast_ac[i] = (stbi__int16) ((k * 256) + (run * 16) + (len + magbits)); + } + } + } +} + +static void stbi__grow_buffer_unsafe(stbi__jpeg *j) +{ + do { + unsigned int b = j->nomore ? 0 : stbi__get8(j->s); + if (b == 0xff) { + int c = stbi__get8(j->s); + while (c == 0xff) c = stbi__get8(j->s); // consume fill bytes + if (c != 0) { + j->marker = (unsigned char) c; + j->nomore = 1; + return; + } + } + j->code_buffer |= b << (24 - j->code_bits); + j->code_bits += 8; + } while (j->code_bits <= 24); +} + +// (1 << n) - 1 +static const stbi__uint32 stbi__bmask[17]={0,1,3,7,15,31,63,127,255,511,1023,2047,4095,8191,16383,32767,65535}; + +// decode a jpeg huffman value from the bitstream +stbi_inline static int stbi__jpeg_huff_decode(stbi__jpeg *j, stbi__huffman *h) +{ + unsigned int temp; + int c,k; + + if (j->code_bits < 16) stbi__grow_buffer_unsafe(j); + + // look at the top FAST_BITS and determine what symbol ID it is, + // if the code is <= FAST_BITS + c = (j->code_buffer >> (32 - FAST_BITS)) & ((1 << FAST_BITS)-1); + k = h->fast[c]; + if (k < 255) { + int s = h->size[k]; + if (s > j->code_bits) + return -1; + j->code_buffer <<= s; + j->code_bits -= s; + return h->values[k]; + } + + // naive test is to shift the code_buffer down so k bits are + // valid, then test against maxcode. To speed this up, we've + // preshifted maxcode left so that it has (16-k) 0s at the + // end; in other words, regardless of the number of bits, it + // wants to be compared against something shifted to have 16; + // that way we don't need to shift inside the loop. + temp = j->code_buffer >> 16; + for (k=FAST_BITS+1 ; ; ++k) + if (temp < h->maxcode[k]) + break; + if (k == 17) { + // error! code not found + j->code_bits -= 16; + return -1; + } + + if (k > j->code_bits) + return -1; + + // convert the huffman code to the symbol id + c = ((j->code_buffer >> (32 - k)) & stbi__bmask[k]) + h->delta[k]; + if(c < 0 || c >= 256) // symbol id out of bounds! + return -1; + STBI_ASSERT((((j->code_buffer) >> (32 - h->size[c])) & stbi__bmask[h->size[c]]) == h->code[c]); + + // convert the id to a symbol + j->code_bits -= k; + j->code_buffer <<= k; + return h->values[c]; +} + +// bias[n] = (-1<code_bits < n) stbi__grow_buffer_unsafe(j); + if (j->code_bits < n) return 0; // ran out of bits from stream, return 0s intead of continuing + + sgn = j->code_buffer >> 31; // sign bit always in MSB; 0 if MSB clear (positive), 1 if MSB set (negative) + k = stbi_lrot(j->code_buffer, n); + j->code_buffer = k & ~stbi__bmask[n]; + k &= stbi__bmask[n]; + j->code_bits -= n; + return k + (stbi__jbias[n] & (sgn - 1)); +} + +// get some unsigned bits +stbi_inline static int stbi__jpeg_get_bits(stbi__jpeg *j, int n) +{ + unsigned int k; + if (j->code_bits < n) stbi__grow_buffer_unsafe(j); + if (j->code_bits < n) return 0; // ran out of bits from stream, return 0s intead of continuing + k = stbi_lrot(j->code_buffer, n); + j->code_buffer = k & ~stbi__bmask[n]; + k &= stbi__bmask[n]; + j->code_bits -= n; + return k; +} + +stbi_inline static int stbi__jpeg_get_bit(stbi__jpeg *j) +{ + unsigned int k; + if (j->code_bits < 1) stbi__grow_buffer_unsafe(j); + if (j->code_bits < 1) return 0; // ran out of bits from stream, return 0s intead of continuing + k = j->code_buffer; + j->code_buffer <<= 1; + --j->code_bits; + return k & 0x80000000; +} + +// given a value that's at position X in the zigzag stream, +// where does it appear in the 8x8 matrix coded as row-major? +static const stbi_uc stbi__jpeg_dezigzag[64+15] = +{ + 0, 1, 8, 16, 9, 2, 3, 10, + 17, 24, 32, 25, 18, 11, 4, 5, + 12, 19, 26, 33, 40, 48, 41, 34, + 27, 20, 13, 6, 7, 14, 21, 28, + 35, 42, 49, 56, 57, 50, 43, 36, + 29, 22, 15, 23, 30, 37, 44, 51, + 58, 59, 52, 45, 38, 31, 39, 46, + 53, 60, 61, 54, 47, 55, 62, 63, + // let corrupt input sample past end + 63, 63, 63, 63, 63, 63, 63, 63, + 63, 63, 63, 63, 63, 63, 63 +}; + +// decode one 64-entry block-- +static int stbi__jpeg_decode_block(stbi__jpeg *j, short data[64], stbi__huffman *hdc, stbi__huffman *hac, stbi__int16 *fac, int b, stbi__uint16 *dequant) +{ + int diff,dc,k; + int t; + + if (j->code_bits < 16) stbi__grow_buffer_unsafe(j); + t = stbi__jpeg_huff_decode(j, hdc); + if (t < 0 || t > 15) return stbi__err("bad huffman code","Corrupt JPEG"); + + // 0 all the ac values now so we can do it 32-bits at a time + memset(data,0,64*sizeof(data[0])); + + diff = t ? stbi__extend_receive(j, t) : 0; + if (!stbi__addints_valid(j->img_comp[b].dc_pred, diff)) return stbi__err("bad delta","Corrupt JPEG"); + dc = j->img_comp[b].dc_pred + diff; + j->img_comp[b].dc_pred = dc; + if (!stbi__mul2shorts_valid(dc, dequant[0])) return stbi__err("can't merge dc and ac", "Corrupt JPEG"); + data[0] = (short) (dc * dequant[0]); + + // decode AC components, see JPEG spec + k = 1; + do { + unsigned int zig; + int c,r,s; + if (j->code_bits < 16) stbi__grow_buffer_unsafe(j); + c = (j->code_buffer >> (32 - FAST_BITS)) & ((1 << FAST_BITS)-1); + r = fac[c]; + if (r) { // fast-AC path + k += (r >> 4) & 15; // run + s = r & 15; // combined length + if (s > j->code_bits) return stbi__err("bad huffman code", "Combined length longer than code bits available"); + j->code_buffer <<= s; + j->code_bits -= s; + // decode into unzigzag'd location + zig = stbi__jpeg_dezigzag[k++]; + data[zig] = (short) ((r >> 8) * dequant[zig]); + } else { + int rs = stbi__jpeg_huff_decode(j, hac); + if (rs < 0) return stbi__err("bad huffman code","Corrupt JPEG"); + s = rs & 15; + r = rs >> 4; + if (s == 0) { + if (rs != 0xf0) break; // end block + k += 16; + } else { + k += r; + // decode into unzigzag'd location + zig = stbi__jpeg_dezigzag[k++]; + data[zig] = (short) (stbi__extend_receive(j,s) * dequant[zig]); + } + } + } while (k < 64); + return 1; +} + +static int stbi__jpeg_decode_block_prog_dc(stbi__jpeg *j, short data[64], stbi__huffman *hdc, int b) +{ + int diff,dc; + int t; + if (j->spec_end != 0) return stbi__err("can't merge dc and ac", "Corrupt JPEG"); + + if (j->code_bits < 16) stbi__grow_buffer_unsafe(j); + + if (j->succ_high == 0) { + // first scan for DC coefficient, must be first + memset(data,0,64*sizeof(data[0])); // 0 all the ac values now + t = stbi__jpeg_huff_decode(j, hdc); + if (t < 0 || t > 15) return stbi__err("can't merge dc and ac", "Corrupt JPEG"); + diff = t ? stbi__extend_receive(j, t) : 0; + + if (!stbi__addints_valid(j->img_comp[b].dc_pred, diff)) return stbi__err("bad delta", "Corrupt JPEG"); + dc = j->img_comp[b].dc_pred + diff; + j->img_comp[b].dc_pred = dc; + if (!stbi__mul2shorts_valid(dc, 1 << j->succ_low)) return stbi__err("can't merge dc and ac", "Corrupt JPEG"); + data[0] = (short) (dc * (1 << j->succ_low)); + } else { + // refinement scan for DC coefficient + if (stbi__jpeg_get_bit(j)) + data[0] += (short) (1 << j->succ_low); + } + return 1; +} + +// @OPTIMIZE: store non-zigzagged during the decode passes, +// and only de-zigzag when dequantizing +static int stbi__jpeg_decode_block_prog_ac(stbi__jpeg *j, short data[64], stbi__huffman *hac, stbi__int16 *fac) +{ + int k; + if (j->spec_start == 0) return stbi__err("can't merge dc and ac", "Corrupt JPEG"); + + if (j->succ_high == 0) { + int shift = j->succ_low; + + if (j->eob_run) { + --j->eob_run; + return 1; + } + + k = j->spec_start; + do { + unsigned int zig; + int c,r,s; + if (j->code_bits < 16) stbi__grow_buffer_unsafe(j); + c = (j->code_buffer >> (32 - FAST_BITS)) & ((1 << FAST_BITS)-1); + r = fac[c]; + if (r) { // fast-AC path + k += (r >> 4) & 15; // run + s = r & 15; // combined length + if (s > j->code_bits) return stbi__err("bad huffman code", "Combined length longer than code bits available"); + j->code_buffer <<= s; + j->code_bits -= s; + zig = stbi__jpeg_dezigzag[k++]; + data[zig] = (short) ((r >> 8) * (1 << shift)); + } else { + int rs = stbi__jpeg_huff_decode(j, hac); + if (rs < 0) return stbi__err("bad huffman code","Corrupt JPEG"); + s = rs & 15; + r = rs >> 4; + if (s == 0) { + if (r < 15) { + j->eob_run = (1 << r); + if (r) + j->eob_run += stbi__jpeg_get_bits(j, r); + --j->eob_run; + break; + } + k += 16; + } else { + k += r; + zig = stbi__jpeg_dezigzag[k++]; + data[zig] = (short) (stbi__extend_receive(j,s) * (1 << shift)); + } + } + } while (k <= j->spec_end); + } else { + // refinement scan for these AC coefficients + + short bit = (short) (1 << j->succ_low); + + if (j->eob_run) { + --j->eob_run; + for (k = j->spec_start; k <= j->spec_end; ++k) { + short *p = &data[stbi__jpeg_dezigzag[k]]; + if (*p != 0) + if (stbi__jpeg_get_bit(j)) + if ((*p & bit)==0) { + if (*p > 0) + *p += bit; + else + *p -= bit; + } + } + } else { + k = j->spec_start; + do { + int r,s; + int rs = stbi__jpeg_huff_decode(j, hac); // @OPTIMIZE see if we can use the fast path here, advance-by-r is so slow, eh + if (rs < 0) return stbi__err("bad huffman code","Corrupt JPEG"); + s = rs & 15; + r = rs >> 4; + if (s == 0) { + if (r < 15) { + j->eob_run = (1 << r) - 1; + if (r) + j->eob_run += stbi__jpeg_get_bits(j, r); + r = 64; // force end of block + } else { + // r=15 s=0 should write 16 0s, so we just do + // a run of 15 0s and then write s (which is 0), + // so we don't have to do anything special here + } + } else { + if (s != 1) return stbi__err("bad huffman code", "Corrupt JPEG"); + // sign bit + if (stbi__jpeg_get_bit(j)) + s = bit; + else + s = -bit; + } + + // advance by r + while (k <= j->spec_end) { + short *p = &data[stbi__jpeg_dezigzag[k++]]; + if (*p != 0) { + if (stbi__jpeg_get_bit(j)) + if ((*p & bit)==0) { + if (*p > 0) + *p += bit; + else + *p -= bit; + } + } else { + if (r == 0) { + *p = (short) s; + break; + } + --r; + } + } + } while (k <= j->spec_end); + } + } + return 1; +} + +// take a -128..127 value and stbi__clamp it and convert to 0..255 +stbi_inline static stbi_uc stbi__clamp(int x) +{ + // trick to use a single test to catch both cases + if ((unsigned int) x > 255) { + if (x < 0) return 0; + if (x > 255) return 255; + } + return (stbi_uc) x; +} + +#define stbi__f2f(x) ((int) (((x) * 4096 + 0.5))) +#define stbi__fsh(x) ((x) * 4096) + +// derived from jidctint -- DCT_ISLOW +#define STBI__IDCT_1D(s0,s1,s2,s3,s4,s5,s6,s7) \ + int t0,t1,t2,t3,p1,p2,p3,p4,p5,x0,x1,x2,x3; \ + p2 = s2; \ + p3 = s6; \ + p1 = (p2+p3) * stbi__f2f(0.5411961f); \ + t2 = p1 + p3*stbi__f2f(-1.847759065f); \ + t3 = p1 + p2*stbi__f2f( 0.765366865f); \ + p2 = s0; \ + p3 = s4; \ + t0 = stbi__fsh(p2+p3); \ + t1 = stbi__fsh(p2-p3); \ + x0 = t0+t3; \ + x3 = t0-t3; \ + x1 = t1+t2; \ + x2 = t1-t2; \ + t0 = s7; \ + t1 = s5; \ + t2 = s3; \ + t3 = s1; \ + p3 = t0+t2; \ + p4 = t1+t3; \ + p1 = t0+t3; \ + p2 = t1+t2; \ + p5 = (p3+p4)*stbi__f2f( 1.175875602f); \ + t0 = t0*stbi__f2f( 0.298631336f); \ + t1 = t1*stbi__f2f( 2.053119869f); \ + t2 = t2*stbi__f2f( 3.072711026f); \ + t3 = t3*stbi__f2f( 1.501321110f); \ + p1 = p5 + p1*stbi__f2f(-0.899976223f); \ + p2 = p5 + p2*stbi__f2f(-2.562915447f); \ + p3 = p3*stbi__f2f(-1.961570560f); \ + p4 = p4*stbi__f2f(-0.390180644f); \ + t3 += p1+p4; \ + t2 += p2+p3; \ + t1 += p2+p4; \ + t0 += p1+p3; + +static void stbi__idct_block(stbi_uc *out, int out_stride, short data[64]) +{ + int i,val[64],*v=val; + stbi_uc *o; + short *d = data; + + // columns + for (i=0; i < 8; ++i,++d, ++v) { + // if all zeroes, shortcut -- this avoids dequantizing 0s and IDCTing + if (d[ 8]==0 && d[16]==0 && d[24]==0 && d[32]==0 + && d[40]==0 && d[48]==0 && d[56]==0) { + // no shortcut 0 seconds + // (1|2|3|4|5|6|7)==0 0 seconds + // all separate -0.047 seconds + // 1 && 2|3 && 4|5 && 6|7: -0.047 seconds + int dcterm = d[0]*4; + v[0] = v[8] = v[16] = v[24] = v[32] = v[40] = v[48] = v[56] = dcterm; + } else { + STBI__IDCT_1D(d[ 0],d[ 8],d[16],d[24],d[32],d[40],d[48],d[56]) + // constants scaled things up by 1<<12; let's bring them back + // down, but keep 2 extra bits of precision + x0 += 512; x1 += 512; x2 += 512; x3 += 512; + v[ 0] = (x0+t3) >> 10; + v[56] = (x0-t3) >> 10; + v[ 8] = (x1+t2) >> 10; + v[48] = (x1-t2) >> 10; + v[16] = (x2+t1) >> 10; + v[40] = (x2-t1) >> 10; + v[24] = (x3+t0) >> 10; + v[32] = (x3-t0) >> 10; + } + } + + for (i=0, v=val, o=out; i < 8; ++i,v+=8,o+=out_stride) { + // no fast case since the first 1D IDCT spread components out + STBI__IDCT_1D(v[0],v[1],v[2],v[3],v[4],v[5],v[6],v[7]) + // constants scaled things up by 1<<12, plus we had 1<<2 from first + // loop, plus horizontal and vertical each scale by sqrt(8) so together + // we've got an extra 1<<3, so 1<<17 total we need to remove. + // so we want to round that, which means adding 0.5 * 1<<17, + // aka 65536. Also, we'll end up with -128 to 127 that we want + // to encode as 0..255 by adding 128, so we'll add that before the shift + x0 += 65536 + (128<<17); + x1 += 65536 + (128<<17); + x2 += 65536 + (128<<17); + x3 += 65536 + (128<<17); + // tried computing the shifts into temps, or'ing the temps to see + // if any were out of range, but that was slower + o[0] = stbi__clamp((x0+t3) >> 17); + o[7] = stbi__clamp((x0-t3) >> 17); + o[1] = stbi__clamp((x1+t2) >> 17); + o[6] = stbi__clamp((x1-t2) >> 17); + o[2] = stbi__clamp((x2+t1) >> 17); + o[5] = stbi__clamp((x2-t1) >> 17); + o[3] = stbi__clamp((x3+t0) >> 17); + o[4] = stbi__clamp((x3-t0) >> 17); + } +} + +#ifdef STBI_SSE2 +// sse2 integer IDCT. not the fastest possible implementation but it +// produces bit-identical results to the generic C version so it's +// fully "transparent". +static void stbi__idct_simd(stbi_uc *out, int out_stride, short data[64]) +{ + // This is constructed to match our regular (generic) integer IDCT exactly. + __m128i row0, row1, row2, row3, row4, row5, row6, row7; + __m128i tmp; + + // dot product constant: even elems=x, odd elems=y + #define dct_const(x,y) _mm_setr_epi16((x),(y),(x),(y),(x),(y),(x),(y)) + + // out(0) = c0[even]*x + c0[odd]*y (c0, x, y 16-bit, out 32-bit) + // out(1) = c1[even]*x + c1[odd]*y + #define dct_rot(out0,out1, x,y,c0,c1) \ + __m128i c0##lo = _mm_unpacklo_epi16((x),(y)); \ + __m128i c0##hi = _mm_unpackhi_epi16((x),(y)); \ + __m128i out0##_l = _mm_madd_epi16(c0##lo, c0); \ + __m128i out0##_h = _mm_madd_epi16(c0##hi, c0); \ + __m128i out1##_l = _mm_madd_epi16(c0##lo, c1); \ + __m128i out1##_h = _mm_madd_epi16(c0##hi, c1) + + // out = in << 12 (in 16-bit, out 32-bit) + #define dct_widen(out, in) \ + __m128i out##_l = _mm_srai_epi32(_mm_unpacklo_epi16(_mm_setzero_si128(), (in)), 4); \ + __m128i out##_h = _mm_srai_epi32(_mm_unpackhi_epi16(_mm_setzero_si128(), (in)), 4) + + // wide add + #define dct_wadd(out, a, b) \ + __m128i out##_l = _mm_add_epi32(a##_l, b##_l); \ + __m128i out##_h = _mm_add_epi32(a##_h, b##_h) + + // wide sub + #define dct_wsub(out, a, b) \ + __m128i out##_l = _mm_sub_epi32(a##_l, b##_l); \ + __m128i out##_h = _mm_sub_epi32(a##_h, b##_h) + + // butterfly a/b, add bias, then shift by "s" and pack + #define dct_bfly32o(out0, out1, a,b,bias,s) \ + { \ + __m128i abiased_l = _mm_add_epi32(a##_l, bias); \ + __m128i abiased_h = _mm_add_epi32(a##_h, bias); \ + dct_wadd(sum, abiased, b); \ + dct_wsub(dif, abiased, b); \ + out0 = _mm_packs_epi32(_mm_srai_epi32(sum_l, s), _mm_srai_epi32(sum_h, s)); \ + out1 = _mm_packs_epi32(_mm_srai_epi32(dif_l, s), _mm_srai_epi32(dif_h, s)); \ + } + + // 8-bit interleave step (for transposes) + #define dct_interleave8(a, b) \ + tmp = a; \ + a = _mm_unpacklo_epi8(a, b); \ + b = _mm_unpackhi_epi8(tmp, b) + + // 16-bit interleave step (for transposes) + #define dct_interleave16(a, b) \ + tmp = a; \ + a = _mm_unpacklo_epi16(a, b); \ + b = _mm_unpackhi_epi16(tmp, b) + + #define dct_pass(bias,shift) \ + { \ + /* even part */ \ + dct_rot(t2e,t3e, row2,row6, rot0_0,rot0_1); \ + __m128i sum04 = _mm_add_epi16(row0, row4); \ + __m128i dif04 = _mm_sub_epi16(row0, row4); \ + dct_widen(t0e, sum04); \ + dct_widen(t1e, dif04); \ + dct_wadd(x0, t0e, t3e); \ + dct_wsub(x3, t0e, t3e); \ + dct_wadd(x1, t1e, t2e); \ + dct_wsub(x2, t1e, t2e); \ + /* odd part */ \ + dct_rot(y0o,y2o, row7,row3, rot2_0,rot2_1); \ + dct_rot(y1o,y3o, row5,row1, rot3_0,rot3_1); \ + __m128i sum17 = _mm_add_epi16(row1, row7); \ + __m128i sum35 = _mm_add_epi16(row3, row5); \ + dct_rot(y4o,y5o, sum17,sum35, rot1_0,rot1_1); \ + dct_wadd(x4, y0o, y4o); \ + dct_wadd(x5, y1o, y5o); \ + dct_wadd(x6, y2o, y5o); \ + dct_wadd(x7, y3o, y4o); \ + dct_bfly32o(row0,row7, x0,x7,bias,shift); \ + dct_bfly32o(row1,row6, x1,x6,bias,shift); \ + dct_bfly32o(row2,row5, x2,x5,bias,shift); \ + dct_bfly32o(row3,row4, x3,x4,bias,shift); \ + } + + __m128i rot0_0 = dct_const(stbi__f2f(0.5411961f), stbi__f2f(0.5411961f) + stbi__f2f(-1.847759065f)); + __m128i rot0_1 = dct_const(stbi__f2f(0.5411961f) + stbi__f2f( 0.765366865f), stbi__f2f(0.5411961f)); + __m128i rot1_0 = dct_const(stbi__f2f(1.175875602f) + stbi__f2f(-0.899976223f), stbi__f2f(1.175875602f)); + __m128i rot1_1 = dct_const(stbi__f2f(1.175875602f), stbi__f2f(1.175875602f) + stbi__f2f(-2.562915447f)); + __m128i rot2_0 = dct_const(stbi__f2f(-1.961570560f) + stbi__f2f( 0.298631336f), stbi__f2f(-1.961570560f)); + __m128i rot2_1 = dct_const(stbi__f2f(-1.961570560f), stbi__f2f(-1.961570560f) + stbi__f2f( 3.072711026f)); + __m128i rot3_0 = dct_const(stbi__f2f(-0.390180644f) + stbi__f2f( 2.053119869f), stbi__f2f(-0.390180644f)); + __m128i rot3_1 = dct_const(stbi__f2f(-0.390180644f), stbi__f2f(-0.390180644f) + stbi__f2f( 1.501321110f)); + + // rounding biases in column/row passes, see stbi__idct_block for explanation. + __m128i bias_0 = _mm_set1_epi32(512); + __m128i bias_1 = _mm_set1_epi32(65536 + (128<<17)); + + // load + row0 = _mm_load_si128((const __m128i *) (data + 0*8)); + row1 = _mm_load_si128((const __m128i *) (data + 1*8)); + row2 = _mm_load_si128((const __m128i *) (data + 2*8)); + row3 = _mm_load_si128((const __m128i *) (data + 3*8)); + row4 = _mm_load_si128((const __m128i *) (data + 4*8)); + row5 = _mm_load_si128((const __m128i *) (data + 5*8)); + row6 = _mm_load_si128((const __m128i *) (data + 6*8)); + row7 = _mm_load_si128((const __m128i *) (data + 7*8)); + + // column pass + dct_pass(bias_0, 10); + + { + // 16bit 8x8 transpose pass 1 + dct_interleave16(row0, row4); + dct_interleave16(row1, row5); + dct_interleave16(row2, row6); + dct_interleave16(row3, row7); + + // transpose pass 2 + dct_interleave16(row0, row2); + dct_interleave16(row1, row3); + dct_interleave16(row4, row6); + dct_interleave16(row5, row7); + + // transpose pass 3 + dct_interleave16(row0, row1); + dct_interleave16(row2, row3); + dct_interleave16(row4, row5); + dct_interleave16(row6, row7); + } + + // row pass + dct_pass(bias_1, 17); + + { + // pack + __m128i p0 = _mm_packus_epi16(row0, row1); // a0a1a2a3...a7b0b1b2b3...b7 + __m128i p1 = _mm_packus_epi16(row2, row3); + __m128i p2 = _mm_packus_epi16(row4, row5); + __m128i p3 = _mm_packus_epi16(row6, row7); + + // 8bit 8x8 transpose pass 1 + dct_interleave8(p0, p2); // a0e0a1e1... + dct_interleave8(p1, p3); // c0g0c1g1... + + // transpose pass 2 + dct_interleave8(p0, p1); // a0c0e0g0... + dct_interleave8(p2, p3); // b0d0f0h0... + + // transpose pass 3 + dct_interleave8(p0, p2); // a0b0c0d0... + dct_interleave8(p1, p3); // a4b4c4d4... + + // store + _mm_storel_epi64((__m128i *) out, p0); out += out_stride; + _mm_storel_epi64((__m128i *) out, _mm_shuffle_epi32(p0, 0x4e)); out += out_stride; + _mm_storel_epi64((__m128i *) out, p2); out += out_stride; + _mm_storel_epi64((__m128i *) out, _mm_shuffle_epi32(p2, 0x4e)); out += out_stride; + _mm_storel_epi64((__m128i *) out, p1); out += out_stride; + _mm_storel_epi64((__m128i *) out, _mm_shuffle_epi32(p1, 0x4e)); out += out_stride; + _mm_storel_epi64((__m128i *) out, p3); out += out_stride; + _mm_storel_epi64((__m128i *) out, _mm_shuffle_epi32(p3, 0x4e)); + } + +#undef dct_const +#undef dct_rot +#undef dct_widen +#undef dct_wadd +#undef dct_wsub +#undef dct_bfly32o +#undef dct_interleave8 +#undef dct_interleave16 +#undef dct_pass +} + +#endif // STBI_SSE2 + +#ifdef STBI_NEON + +// NEON integer IDCT. should produce bit-identical +// results to the generic C version. +static void stbi__idct_simd(stbi_uc *out, int out_stride, short data[64]) +{ + int16x8_t row0, row1, row2, row3, row4, row5, row6, row7; + + int16x4_t rot0_0 = vdup_n_s16(stbi__f2f(0.5411961f)); + int16x4_t rot0_1 = vdup_n_s16(stbi__f2f(-1.847759065f)); + int16x4_t rot0_2 = vdup_n_s16(stbi__f2f( 0.765366865f)); + int16x4_t rot1_0 = vdup_n_s16(stbi__f2f( 1.175875602f)); + int16x4_t rot1_1 = vdup_n_s16(stbi__f2f(-0.899976223f)); + int16x4_t rot1_2 = vdup_n_s16(stbi__f2f(-2.562915447f)); + int16x4_t rot2_0 = vdup_n_s16(stbi__f2f(-1.961570560f)); + int16x4_t rot2_1 = vdup_n_s16(stbi__f2f(-0.390180644f)); + int16x4_t rot3_0 = vdup_n_s16(stbi__f2f( 0.298631336f)); + int16x4_t rot3_1 = vdup_n_s16(stbi__f2f( 2.053119869f)); + int16x4_t rot3_2 = vdup_n_s16(stbi__f2f( 3.072711026f)); + int16x4_t rot3_3 = vdup_n_s16(stbi__f2f( 1.501321110f)); + +#define dct_long_mul(out, inq, coeff) \ + int32x4_t out##_l = vmull_s16(vget_low_s16(inq), coeff); \ + int32x4_t out##_h = vmull_s16(vget_high_s16(inq), coeff) + +#define dct_long_mac(out, acc, inq, coeff) \ + int32x4_t out##_l = vmlal_s16(acc##_l, vget_low_s16(inq), coeff); \ + int32x4_t out##_h = vmlal_s16(acc##_h, vget_high_s16(inq), coeff) + +#define dct_widen(out, inq) \ + int32x4_t out##_l = vshll_n_s16(vget_low_s16(inq), 12); \ + int32x4_t out##_h = vshll_n_s16(vget_high_s16(inq), 12) + +// wide add +#define dct_wadd(out, a, b) \ + int32x4_t out##_l = vaddq_s32(a##_l, b##_l); \ + int32x4_t out##_h = vaddq_s32(a##_h, b##_h) + +// wide sub +#define dct_wsub(out, a, b) \ + int32x4_t out##_l = vsubq_s32(a##_l, b##_l); \ + int32x4_t out##_h = vsubq_s32(a##_h, b##_h) + +// butterfly a/b, then shift using "shiftop" by "s" and pack +#define dct_bfly32o(out0,out1, a,b,shiftop,s) \ + { \ + dct_wadd(sum, a, b); \ + dct_wsub(dif, a, b); \ + out0 = vcombine_s16(shiftop(sum_l, s), shiftop(sum_h, s)); \ + out1 = vcombine_s16(shiftop(dif_l, s), shiftop(dif_h, s)); \ + } + +#define dct_pass(shiftop, shift) \ + { \ + /* even part */ \ + int16x8_t sum26 = vaddq_s16(row2, row6); \ + dct_long_mul(p1e, sum26, rot0_0); \ + dct_long_mac(t2e, p1e, row6, rot0_1); \ + dct_long_mac(t3e, p1e, row2, rot0_2); \ + int16x8_t sum04 = vaddq_s16(row0, row4); \ + int16x8_t dif04 = vsubq_s16(row0, row4); \ + dct_widen(t0e, sum04); \ + dct_widen(t1e, dif04); \ + dct_wadd(x0, t0e, t3e); \ + dct_wsub(x3, t0e, t3e); \ + dct_wadd(x1, t1e, t2e); \ + dct_wsub(x2, t1e, t2e); \ + /* odd part */ \ + int16x8_t sum15 = vaddq_s16(row1, row5); \ + int16x8_t sum17 = vaddq_s16(row1, row7); \ + int16x8_t sum35 = vaddq_s16(row3, row5); \ + int16x8_t sum37 = vaddq_s16(row3, row7); \ + int16x8_t sumodd = vaddq_s16(sum17, sum35); \ + dct_long_mul(p5o, sumodd, rot1_0); \ + dct_long_mac(p1o, p5o, sum17, rot1_1); \ + dct_long_mac(p2o, p5o, sum35, rot1_2); \ + dct_long_mul(p3o, sum37, rot2_0); \ + dct_long_mul(p4o, sum15, rot2_1); \ + dct_wadd(sump13o, p1o, p3o); \ + dct_wadd(sump24o, p2o, p4o); \ + dct_wadd(sump23o, p2o, p3o); \ + dct_wadd(sump14o, p1o, p4o); \ + dct_long_mac(x4, sump13o, row7, rot3_0); \ + dct_long_mac(x5, sump24o, row5, rot3_1); \ + dct_long_mac(x6, sump23o, row3, rot3_2); \ + dct_long_mac(x7, sump14o, row1, rot3_3); \ + dct_bfly32o(row0,row7, x0,x7,shiftop,shift); \ + dct_bfly32o(row1,row6, x1,x6,shiftop,shift); \ + dct_bfly32o(row2,row5, x2,x5,shiftop,shift); \ + dct_bfly32o(row3,row4, x3,x4,shiftop,shift); \ + } + + // load + row0 = vld1q_s16(data + 0*8); + row1 = vld1q_s16(data + 1*8); + row2 = vld1q_s16(data + 2*8); + row3 = vld1q_s16(data + 3*8); + row4 = vld1q_s16(data + 4*8); + row5 = vld1q_s16(data + 5*8); + row6 = vld1q_s16(data + 6*8); + row7 = vld1q_s16(data + 7*8); + + // add DC bias + row0 = vaddq_s16(row0, vsetq_lane_s16(1024, vdupq_n_s16(0), 0)); + + // column pass + dct_pass(vrshrn_n_s32, 10); + + // 16bit 8x8 transpose + { +// these three map to a single VTRN.16, VTRN.32, and VSWP, respectively. +// whether compilers actually get this is another story, sadly. +#define dct_trn16(x, y) { int16x8x2_t t = vtrnq_s16(x, y); x = t.val[0]; y = t.val[1]; } +#define dct_trn32(x, y) { int32x4x2_t t = vtrnq_s32(vreinterpretq_s32_s16(x), vreinterpretq_s32_s16(y)); x = vreinterpretq_s16_s32(t.val[0]); y = vreinterpretq_s16_s32(t.val[1]); } +#define dct_trn64(x, y) { int16x8_t x0 = x; int16x8_t y0 = y; x = vcombine_s16(vget_low_s16(x0), vget_low_s16(y0)); y = vcombine_s16(vget_high_s16(x0), vget_high_s16(y0)); } + + // pass 1 + dct_trn16(row0, row1); // a0b0a2b2a4b4a6b6 + dct_trn16(row2, row3); + dct_trn16(row4, row5); + dct_trn16(row6, row7); + + // pass 2 + dct_trn32(row0, row2); // a0b0c0d0a4b4c4d4 + dct_trn32(row1, row3); + dct_trn32(row4, row6); + dct_trn32(row5, row7); + + // pass 3 + dct_trn64(row0, row4); // a0b0c0d0e0f0g0h0 + dct_trn64(row1, row5); + dct_trn64(row2, row6); + dct_trn64(row3, row7); + +#undef dct_trn16 +#undef dct_trn32 +#undef dct_trn64 + } + + // row pass + // vrshrn_n_s32 only supports shifts up to 16, we need + // 17. so do a non-rounding shift of 16 first then follow + // up with a rounding shift by 1. + dct_pass(vshrn_n_s32, 16); + + { + // pack and round + uint8x8_t p0 = vqrshrun_n_s16(row0, 1); + uint8x8_t p1 = vqrshrun_n_s16(row1, 1); + uint8x8_t p2 = vqrshrun_n_s16(row2, 1); + uint8x8_t p3 = vqrshrun_n_s16(row3, 1); + uint8x8_t p4 = vqrshrun_n_s16(row4, 1); + uint8x8_t p5 = vqrshrun_n_s16(row5, 1); + uint8x8_t p6 = vqrshrun_n_s16(row6, 1); + uint8x8_t p7 = vqrshrun_n_s16(row7, 1); + + // again, these can translate into one instruction, but often don't. +#define dct_trn8_8(x, y) { uint8x8x2_t t = vtrn_u8(x, y); x = t.val[0]; y = t.val[1]; } +#define dct_trn8_16(x, y) { uint16x4x2_t t = vtrn_u16(vreinterpret_u16_u8(x), vreinterpret_u16_u8(y)); x = vreinterpret_u8_u16(t.val[0]); y = vreinterpret_u8_u16(t.val[1]); } +#define dct_trn8_32(x, y) { uint32x2x2_t t = vtrn_u32(vreinterpret_u32_u8(x), vreinterpret_u32_u8(y)); x = vreinterpret_u8_u32(t.val[0]); y = vreinterpret_u8_u32(t.val[1]); } + + // sadly can't use interleaved stores here since we only write + // 8 bytes to each scan line! + + // 8x8 8-bit transpose pass 1 + dct_trn8_8(p0, p1); + dct_trn8_8(p2, p3); + dct_trn8_8(p4, p5); + dct_trn8_8(p6, p7); + + // pass 2 + dct_trn8_16(p0, p2); + dct_trn8_16(p1, p3); + dct_trn8_16(p4, p6); + dct_trn8_16(p5, p7); + + // pass 3 + dct_trn8_32(p0, p4); + dct_trn8_32(p1, p5); + dct_trn8_32(p2, p6); + dct_trn8_32(p3, p7); + + // store + vst1_u8(out, p0); out += out_stride; + vst1_u8(out, p1); out += out_stride; + vst1_u8(out, p2); out += out_stride; + vst1_u8(out, p3); out += out_stride; + vst1_u8(out, p4); out += out_stride; + vst1_u8(out, p5); out += out_stride; + vst1_u8(out, p6); out += out_stride; + vst1_u8(out, p7); + +#undef dct_trn8_8 +#undef dct_trn8_16 +#undef dct_trn8_32 + } + +#undef dct_long_mul +#undef dct_long_mac +#undef dct_widen +#undef dct_wadd +#undef dct_wsub +#undef dct_bfly32o +#undef dct_pass +} + +#endif // STBI_NEON + +#define STBI__MARKER_none 0xff +// if there's a pending marker from the entropy stream, return that +// otherwise, fetch from the stream and get a marker. if there's no +// marker, return 0xff, which is never a valid marker value +static stbi_uc stbi__get_marker(stbi__jpeg *j) +{ + stbi_uc x; + if (j->marker != STBI__MARKER_none) { x = j->marker; j->marker = STBI__MARKER_none; return x; } + x = stbi__get8(j->s); + if (x != 0xff) return STBI__MARKER_none; + while (x == 0xff) + x = stbi__get8(j->s); // consume repeated 0xff fill bytes + return x; +} + +// in each scan, we'll have scan_n components, and the order +// of the components is specified by order[] +#define STBI__RESTART(x) ((x) >= 0xd0 && (x) <= 0xd7) + +// after a restart interval, stbi__jpeg_reset the entropy decoder and +// the dc prediction +static void stbi__jpeg_reset(stbi__jpeg *j) +{ + j->code_bits = 0; + j->code_buffer = 0; + j->nomore = 0; + j->img_comp[0].dc_pred = j->img_comp[1].dc_pred = j->img_comp[2].dc_pred = j->img_comp[3].dc_pred = 0; + j->marker = STBI__MARKER_none; + j->todo = j->restart_interval ? j->restart_interval : 0x7fffffff; + j->eob_run = 0; + // no more than 1<<31 MCUs if no restart_interal? that's plenty safe, + // since we don't even allow 1<<30 pixels +} + +static int stbi__parse_entropy_coded_data(stbi__jpeg *z) +{ + stbi__jpeg_reset(z); + if (!z->progressive) { + if (z->scan_n == 1) { + int i,j; + STBI_SIMD_ALIGN(short, data[64]); + int n = z->order[0]; + // non-interleaved data, we just need to process one block at a time, + // in trivial scanline order + // number of blocks to do just depends on how many actual "pixels" this + // component has, independent of interleaved MCU blocking and such + int w = (z->img_comp[n].x+7) >> 3; + int h = (z->img_comp[n].y+7) >> 3; + for (j=0; j < h; ++j) { + for (i=0; i < w; ++i) { + int ha = z->img_comp[n].ha; + if (!stbi__jpeg_decode_block(z, data, z->huff_dc+z->img_comp[n].hd, z->huff_ac+ha, z->fast_ac[ha], n, z->dequant[z->img_comp[n].tq])) return 0; + z->idct_block_kernel(z->img_comp[n].data+z->img_comp[n].w2*j*8+i*8, z->img_comp[n].w2, data); + // every data block is an MCU, so countdown the restart interval + if (--z->todo <= 0) { + if (z->code_bits < 24) stbi__grow_buffer_unsafe(z); + // if it's NOT a restart, then just bail, so we get corrupt data + // rather than no data + if (!STBI__RESTART(z->marker)) return 1; + stbi__jpeg_reset(z); + } + } + } + return 1; + } else { // interleaved + int i,j,k,x,y; + STBI_SIMD_ALIGN(short, data[64]); + for (j=0; j < z->img_mcu_y; ++j) { + for (i=0; i < z->img_mcu_x; ++i) { + // scan an interleaved mcu... process scan_n components in order + for (k=0; k < z->scan_n; ++k) { + int n = z->order[k]; + // scan out an mcu's worth of this component; that's just determined + // by the basic H and V specified for the component + for (y=0; y < z->img_comp[n].v; ++y) { + for (x=0; x < z->img_comp[n].h; ++x) { + int x2 = (i*z->img_comp[n].h + x)*8; + int y2 = (j*z->img_comp[n].v + y)*8; + int ha = z->img_comp[n].ha; + if (!stbi__jpeg_decode_block(z, data, z->huff_dc+z->img_comp[n].hd, z->huff_ac+ha, z->fast_ac[ha], n, z->dequant[z->img_comp[n].tq])) return 0; + z->idct_block_kernel(z->img_comp[n].data+z->img_comp[n].w2*y2+x2, z->img_comp[n].w2, data); + } + } + } + // after all interleaved components, that's an interleaved MCU, + // so now count down the restart interval + if (--z->todo <= 0) { + if (z->code_bits < 24) stbi__grow_buffer_unsafe(z); + if (!STBI__RESTART(z->marker)) return 1; + stbi__jpeg_reset(z); + } + } + } + return 1; + } + } else { + if (z->scan_n == 1) { + int i,j; + int n = z->order[0]; + // non-interleaved data, we just need to process one block at a time, + // in trivial scanline order + // number of blocks to do just depends on how many actual "pixels" this + // component has, independent of interleaved MCU blocking and such + int w = (z->img_comp[n].x+7) >> 3; + int h = (z->img_comp[n].y+7) >> 3; + for (j=0; j < h; ++j) { + for (i=0; i < w; ++i) { + short *data = z->img_comp[n].coeff + 64 * (i + j * z->img_comp[n].coeff_w); + if (z->spec_start == 0) { + if (!stbi__jpeg_decode_block_prog_dc(z, data, &z->huff_dc[z->img_comp[n].hd], n)) + return 0; + } else { + int ha = z->img_comp[n].ha; + if (!stbi__jpeg_decode_block_prog_ac(z, data, &z->huff_ac[ha], z->fast_ac[ha])) + return 0; + } + // every data block is an MCU, so countdown the restart interval + if (--z->todo <= 0) { + if (z->code_bits < 24) stbi__grow_buffer_unsafe(z); + if (!STBI__RESTART(z->marker)) return 1; + stbi__jpeg_reset(z); + } + } + } + return 1; + } else { // interleaved + int i,j,k,x,y; + for (j=0; j < z->img_mcu_y; ++j) { + for (i=0; i < z->img_mcu_x; ++i) { + // scan an interleaved mcu... process scan_n components in order + for (k=0; k < z->scan_n; ++k) { + int n = z->order[k]; + // scan out an mcu's worth of this component; that's just determined + // by the basic H and V specified for the component + for (y=0; y < z->img_comp[n].v; ++y) { + for (x=0; x < z->img_comp[n].h; ++x) { + int x2 = (i*z->img_comp[n].h + x); + int y2 = (j*z->img_comp[n].v + y); + short *data = z->img_comp[n].coeff + 64 * (x2 + y2 * z->img_comp[n].coeff_w); + if (!stbi__jpeg_decode_block_prog_dc(z, data, &z->huff_dc[z->img_comp[n].hd], n)) + return 0; + } + } + } + // after all interleaved components, that's an interleaved MCU, + // so now count down the restart interval + if (--z->todo <= 0) { + if (z->code_bits < 24) stbi__grow_buffer_unsafe(z); + if (!STBI__RESTART(z->marker)) return 1; + stbi__jpeg_reset(z); + } + } + } + return 1; + } + } +} + +static void stbi__jpeg_dequantize(short *data, stbi__uint16 *dequant) +{ + int i; + for (i=0; i < 64; ++i) + data[i] *= dequant[i]; +} + +static void stbi__jpeg_finish(stbi__jpeg *z) +{ + if (z->progressive) { + // dequantize and idct the data + int i,j,n; + for (n=0; n < z->s->img_n; ++n) { + int w = (z->img_comp[n].x+7) >> 3; + int h = (z->img_comp[n].y+7) >> 3; + for (j=0; j < h; ++j) { + for (i=0; i < w; ++i) { + short *data = z->img_comp[n].coeff + 64 * (i + j * z->img_comp[n].coeff_w); + stbi__jpeg_dequantize(data, z->dequant[z->img_comp[n].tq]); + z->idct_block_kernel(z->img_comp[n].data+z->img_comp[n].w2*j*8+i*8, z->img_comp[n].w2, data); + } + } + } + } +} + +static int stbi__process_marker(stbi__jpeg *z, int m) +{ + int L; + switch (m) { + case STBI__MARKER_none: // no marker found + return stbi__err("expected marker","Corrupt JPEG"); + + case 0xDD: // DRI - specify restart interval + if (stbi__get16be(z->s) != 4) return stbi__err("bad DRI len","Corrupt JPEG"); + z->restart_interval = stbi__get16be(z->s); + return 1; + + case 0xDB: // DQT - define quantization table + L = stbi__get16be(z->s)-2; + while (L > 0) { + int q = stbi__get8(z->s); + int p = q >> 4, sixteen = (p != 0); + int t = q & 15,i; + if (p != 0 && p != 1) return stbi__err("bad DQT type","Corrupt JPEG"); + if (t > 3) return stbi__err("bad DQT table","Corrupt JPEG"); + + for (i=0; i < 64; ++i) + z->dequant[t][stbi__jpeg_dezigzag[i]] = (stbi__uint16)(sixteen ? stbi__get16be(z->s) : stbi__get8(z->s)); + L -= (sixteen ? 129 : 65); + } + return L==0; + + case 0xC4: // DHT - define huffman table + L = stbi__get16be(z->s)-2; + while (L > 0) { + stbi_uc *v; + int sizes[16],i,n=0; + int q = stbi__get8(z->s); + int tc = q >> 4; + int th = q & 15; + if (tc > 1 || th > 3) return stbi__err("bad DHT header","Corrupt JPEG"); + for (i=0; i < 16; ++i) { + sizes[i] = stbi__get8(z->s); + n += sizes[i]; + } + if(n > 256) return stbi__err("bad DHT header","Corrupt JPEG"); // Loop over i < n would write past end of values! + L -= 17; + if (tc == 0) { + if (!stbi__build_huffman(z->huff_dc+th, sizes)) return 0; + v = z->huff_dc[th].values; + } else { + if (!stbi__build_huffman(z->huff_ac+th, sizes)) return 0; + v = z->huff_ac[th].values; + } + for (i=0; i < n; ++i) + v[i] = stbi__get8(z->s); + if (tc != 0) + stbi__build_fast_ac(z->fast_ac[th], z->huff_ac + th); + L -= n; + } + return L==0; + } + + // check for comment block or APP blocks + if ((m >= 0xE0 && m <= 0xEF) || m == 0xFE) { + L = stbi__get16be(z->s); + if (L < 2) { + if (m == 0xFE) + return stbi__err("bad COM len","Corrupt JPEG"); + else + return stbi__err("bad APP len","Corrupt JPEG"); + } + L -= 2; + + if (m == 0xE0 && L >= 5) { // JFIF APP0 segment + static const unsigned char tag[5] = {'J','F','I','F','\0'}; + int ok = 1; + int i; + for (i=0; i < 5; ++i) + if (stbi__get8(z->s) != tag[i]) + ok = 0; + L -= 5; + if (ok) + z->jfif = 1; + } else if (m == 0xEE && L >= 12) { // Adobe APP14 segment + static const unsigned char tag[6] = {'A','d','o','b','e','\0'}; + int ok = 1; + int i; + for (i=0; i < 6; ++i) + if (stbi__get8(z->s) != tag[i]) + ok = 0; + L -= 6; + if (ok) { + stbi__get8(z->s); // version + stbi__get16be(z->s); // flags0 + stbi__get16be(z->s); // flags1 + z->app14_color_transform = stbi__get8(z->s); // color transform + L -= 6; + } + } + + stbi__skip(z->s, L); + return 1; + } + + return stbi__err("unknown marker","Corrupt JPEG"); +} + +// after we see SOS +static int stbi__process_scan_header(stbi__jpeg *z) +{ + int i; + int Ls = stbi__get16be(z->s); + z->scan_n = stbi__get8(z->s); + if (z->scan_n < 1 || z->scan_n > 4 || z->scan_n > (int) z->s->img_n) return stbi__err("bad SOS component count","Corrupt JPEG"); + if (Ls != 6+2*z->scan_n) return stbi__err("bad SOS len","Corrupt JPEG"); + for (i=0; i < z->scan_n; ++i) { + int id = stbi__get8(z->s), which; + int q = stbi__get8(z->s); + for (which = 0; which < z->s->img_n; ++which) + if (z->img_comp[which].id == id) + break; + if (which == z->s->img_n) return 0; // no match + z->img_comp[which].hd = q >> 4; if (z->img_comp[which].hd > 3) return stbi__err("bad DC huff","Corrupt JPEG"); + z->img_comp[which].ha = q & 15; if (z->img_comp[which].ha > 3) return stbi__err("bad AC huff","Corrupt JPEG"); + z->order[i] = which; + } + + { + int aa; + z->spec_start = stbi__get8(z->s); + z->spec_end = stbi__get8(z->s); // should be 63, but might be 0 + aa = stbi__get8(z->s); + z->succ_high = (aa >> 4); + z->succ_low = (aa & 15); + if (z->progressive) { + if (z->spec_start > 63 || z->spec_end > 63 || z->spec_start > z->spec_end || z->succ_high > 13 || z->succ_low > 13) + return stbi__err("bad SOS", "Corrupt JPEG"); + } else { + if (z->spec_start != 0) return stbi__err("bad SOS","Corrupt JPEG"); + if (z->succ_high != 0 || z->succ_low != 0) return stbi__err("bad SOS","Corrupt JPEG"); + z->spec_end = 63; + } + } + + return 1; +} + +static int stbi__free_jpeg_components(stbi__jpeg *z, int ncomp, int why) +{ + int i; + for (i=0; i < ncomp; ++i) { + if (z->img_comp[i].raw_data) { + STBI_FREE(z->img_comp[i].raw_data); + z->img_comp[i].raw_data = NULL; + z->img_comp[i].data = NULL; + } + if (z->img_comp[i].raw_coeff) { + STBI_FREE(z->img_comp[i].raw_coeff); + z->img_comp[i].raw_coeff = 0; + z->img_comp[i].coeff = 0; + } + if (z->img_comp[i].linebuf) { + STBI_FREE(z->img_comp[i].linebuf); + z->img_comp[i].linebuf = NULL; + } + } + return why; +} + +static int stbi__process_frame_header(stbi__jpeg *z, int scan) +{ + stbi__context *s = z->s; + int Lf,p,i,q, h_max=1,v_max=1,c; + Lf = stbi__get16be(s); if (Lf < 11) return stbi__err("bad SOF len","Corrupt JPEG"); // JPEG + p = stbi__get8(s); if (p != 8) return stbi__err("only 8-bit","JPEG format not supported: 8-bit only"); // JPEG baseline + s->img_y = stbi__get16be(s); if (s->img_y == 0) return stbi__err("no header height", "JPEG format not supported: delayed height"); // Legal, but we don't handle it--but neither does IJG + s->img_x = stbi__get16be(s); if (s->img_x == 0) return stbi__err("0 width","Corrupt JPEG"); // JPEG requires + if (s->img_y > STBI_MAX_DIMENSIONS) return stbi__err("too large","Very large image (corrupt?)"); + if (s->img_x > STBI_MAX_DIMENSIONS) return stbi__err("too large","Very large image (corrupt?)"); + c = stbi__get8(s); + if (c != 3 && c != 1 && c != 4) return stbi__err("bad component count","Corrupt JPEG"); + s->img_n = c; + for (i=0; i < c; ++i) { + z->img_comp[i].data = NULL; + z->img_comp[i].linebuf = NULL; + } + + if (Lf != 8+3*s->img_n) return stbi__err("bad SOF len","Corrupt JPEG"); + + z->rgb = 0; + for (i=0; i < s->img_n; ++i) { + static const unsigned char rgb[3] = { 'R', 'G', 'B' }; + z->img_comp[i].id = stbi__get8(s); + if (s->img_n == 3 && z->img_comp[i].id == rgb[i]) + ++z->rgb; + q = stbi__get8(s); + z->img_comp[i].h = (q >> 4); if (!z->img_comp[i].h || z->img_comp[i].h > 4) return stbi__err("bad H","Corrupt JPEG"); + z->img_comp[i].v = q & 15; if (!z->img_comp[i].v || z->img_comp[i].v > 4) return stbi__err("bad V","Corrupt JPEG"); + z->img_comp[i].tq = stbi__get8(s); if (z->img_comp[i].tq > 3) return stbi__err("bad TQ","Corrupt JPEG"); + } + + if (scan != STBI__SCAN_load) return 1; + + if (!stbi__mad3sizes_valid(s->img_x, s->img_y, s->img_n, 0)) return stbi__err("too large", "Image too large to decode"); + + for (i=0; i < s->img_n; ++i) { + if (z->img_comp[i].h > h_max) h_max = z->img_comp[i].h; + if (z->img_comp[i].v > v_max) v_max = z->img_comp[i].v; + } + + // check that plane subsampling factors are integer ratios; our resamplers can't deal with fractional ratios + // and I've never seen a non-corrupted JPEG file actually use them + for (i=0; i < s->img_n; ++i) { + if (h_max % z->img_comp[i].h != 0) return stbi__err("bad H","Corrupt JPEG"); + if (v_max % z->img_comp[i].v != 0) return stbi__err("bad V","Corrupt JPEG"); + } + + // compute interleaved mcu info + z->img_h_max = h_max; + z->img_v_max = v_max; + z->img_mcu_w = h_max * 8; + z->img_mcu_h = v_max * 8; + // these sizes can't be more than 17 bits + z->img_mcu_x = (s->img_x + z->img_mcu_w-1) / z->img_mcu_w; + z->img_mcu_y = (s->img_y + z->img_mcu_h-1) / z->img_mcu_h; + + for (i=0; i < s->img_n; ++i) { + // number of effective pixels (e.g. for non-interleaved MCU) + z->img_comp[i].x = (s->img_x * z->img_comp[i].h + h_max-1) / h_max; + z->img_comp[i].y = (s->img_y * z->img_comp[i].v + v_max-1) / v_max; + // to simplify generation, we'll allocate enough memory to decode + // the bogus oversized data from using interleaved MCUs and their + // big blocks (e.g. a 16x16 iMCU on an image of width 33); we won't + // discard the extra data until colorspace conversion + // + // img_mcu_x, img_mcu_y: <=17 bits; comp[i].h and .v are <=4 (checked earlier) + // so these muls can't overflow with 32-bit ints (which we require) + z->img_comp[i].w2 = z->img_mcu_x * z->img_comp[i].h * 8; + z->img_comp[i].h2 = z->img_mcu_y * z->img_comp[i].v * 8; + z->img_comp[i].coeff = 0; + z->img_comp[i].raw_coeff = 0; + z->img_comp[i].linebuf = NULL; + z->img_comp[i].raw_data = stbi__malloc_mad2(z->img_comp[i].w2, z->img_comp[i].h2, 15); + if (z->img_comp[i].raw_data == NULL) + return stbi__free_jpeg_components(z, i+1, stbi__err("outofmem", "Out of memory")); + // align blocks for idct using mmx/sse + z->img_comp[i].data = (stbi_uc*) (((size_t) z->img_comp[i].raw_data + 15) & ~15); + if (z->progressive) { + // w2, h2 are multiples of 8 (see above) + z->img_comp[i].coeff_w = z->img_comp[i].w2 / 8; + z->img_comp[i].coeff_h = z->img_comp[i].h2 / 8; + z->img_comp[i].raw_coeff = stbi__malloc_mad3(z->img_comp[i].w2, z->img_comp[i].h2, sizeof(short), 15); + if (z->img_comp[i].raw_coeff == NULL) + return stbi__free_jpeg_components(z, i+1, stbi__err("outofmem", "Out of memory")); + z->img_comp[i].coeff = (short*) (((size_t) z->img_comp[i].raw_coeff + 15) & ~15); + } + } + + return 1; +} + +// use comparisons since in some cases we handle more than one case (e.g. SOF) +#define stbi__DNL(x) ((x) == 0xdc) +#define stbi__SOI(x) ((x) == 0xd8) +#define stbi__EOI(x) ((x) == 0xd9) +#define stbi__SOF(x) ((x) == 0xc0 || (x) == 0xc1 || (x) == 0xc2) +#define stbi__SOS(x) ((x) == 0xda) + +#define stbi__SOF_progressive(x) ((x) == 0xc2) + +static int stbi__decode_jpeg_header(stbi__jpeg *z, int scan) +{ + int m; + z->jfif = 0; + z->app14_color_transform = -1; // valid values are 0,1,2 + z->marker = STBI__MARKER_none; // initialize cached marker to empty + m = stbi__get_marker(z); + if (!stbi__SOI(m)) return stbi__err("no SOI","Corrupt JPEG"); + if (scan == STBI__SCAN_type) return 1; + m = stbi__get_marker(z); + while (!stbi__SOF(m)) { + if (!stbi__process_marker(z,m)) return 0; + m = stbi__get_marker(z); + while (m == STBI__MARKER_none) { + // some files have extra padding after their blocks, so ok, we'll scan + if (stbi__at_eof(z->s)) return stbi__err("no SOF", "Corrupt JPEG"); + m = stbi__get_marker(z); + } + } + z->progressive = stbi__SOF_progressive(m); + if (!stbi__process_frame_header(z, scan)) return 0; + return 1; +} + +static stbi_uc stbi__skip_jpeg_junk_at_end(stbi__jpeg *j) +{ + // some JPEGs have junk at end, skip over it but if we find what looks + // like a valid marker, resume there + while (!stbi__at_eof(j->s)) { + stbi_uc x = stbi__get8(j->s); + while (x == 0xff) { // might be a marker + if (stbi__at_eof(j->s)) return STBI__MARKER_none; + x = stbi__get8(j->s); + if (x != 0x00 && x != 0xff) { + // not a stuffed zero or lead-in to another marker, looks + // like an actual marker, return it + return x; + } + // stuffed zero has x=0 now which ends the loop, meaning we go + // back to regular scan loop. + // repeated 0xff keeps trying to read the next byte of the marker. + } + } + return STBI__MARKER_none; +} + +// decode image to YCbCr format +static int stbi__decode_jpeg_image(stbi__jpeg *j) +{ + int m; + for (m = 0; m < 4; m++) { + j->img_comp[m].raw_data = NULL; + j->img_comp[m].raw_coeff = NULL; + } + j->restart_interval = 0; + if (!stbi__decode_jpeg_header(j, STBI__SCAN_load)) return 0; + m = stbi__get_marker(j); + while (!stbi__EOI(m)) { + if (stbi__SOS(m)) { + if (!stbi__process_scan_header(j)) return 0; + if (!stbi__parse_entropy_coded_data(j)) return 0; + if (j->marker == STBI__MARKER_none ) { + j->marker = stbi__skip_jpeg_junk_at_end(j); + // if we reach eof without hitting a marker, stbi__get_marker() below will fail and we'll eventually return 0 + } + m = stbi__get_marker(j); + if (STBI__RESTART(m)) + m = stbi__get_marker(j); + } else if (stbi__DNL(m)) { + int Ld = stbi__get16be(j->s); + stbi__uint32 NL = stbi__get16be(j->s); + if (Ld != 4) return stbi__err("bad DNL len", "Corrupt JPEG"); + if (NL != j->s->img_y) return stbi__err("bad DNL height", "Corrupt JPEG"); + m = stbi__get_marker(j); + } else { + if (!stbi__process_marker(j, m)) return 1; + m = stbi__get_marker(j); + } + } + if (j->progressive) + stbi__jpeg_finish(j); + return 1; +} + +// static jfif-centered resampling (across block boundaries) + +typedef stbi_uc *(*resample_row_func)(stbi_uc *out, stbi_uc *in0, stbi_uc *in1, + int w, int hs); + +#define stbi__div4(x) ((stbi_uc) ((x) >> 2)) + +static stbi_uc *resample_row_1(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs) +{ + STBI_NOTUSED(out); + STBI_NOTUSED(in_far); + STBI_NOTUSED(w); + STBI_NOTUSED(hs); + return in_near; +} + +static stbi_uc* stbi__resample_row_v_2(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs) +{ + // need to generate two samples vertically for every one in input + int i; + STBI_NOTUSED(hs); + for (i=0; i < w; ++i) + out[i] = stbi__div4(3*in_near[i] + in_far[i] + 2); + return out; +} + +static stbi_uc* stbi__resample_row_h_2(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs) +{ + // need to generate two samples horizontally for every one in input + int i; + stbi_uc *input = in_near; + + if (w == 1) { + // if only one sample, can't do any interpolation + out[0] = out[1] = input[0]; + return out; + } + + out[0] = input[0]; + out[1] = stbi__div4(input[0]*3 + input[1] + 2); + for (i=1; i < w-1; ++i) { + int n = 3*input[i]+2; + out[i*2+0] = stbi__div4(n+input[i-1]); + out[i*2+1] = stbi__div4(n+input[i+1]); + } + out[i*2+0] = stbi__div4(input[w-2]*3 + input[w-1] + 2); + out[i*2+1] = input[w-1]; + + STBI_NOTUSED(in_far); + STBI_NOTUSED(hs); + + return out; +} + +#define stbi__div16(x) ((stbi_uc) ((x) >> 4)) + +static stbi_uc *stbi__resample_row_hv_2(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs) +{ + // need to generate 2x2 samples for every one in input + int i,t0,t1; + if (w == 1) { + out[0] = out[1] = stbi__div4(3*in_near[0] + in_far[0] + 2); + return out; + } + + t1 = 3*in_near[0] + in_far[0]; + out[0] = stbi__div4(t1+2); + for (i=1; i < w; ++i) { + t0 = t1; + t1 = 3*in_near[i]+in_far[i]; + out[i*2-1] = stbi__div16(3*t0 + t1 + 8); + out[i*2 ] = stbi__div16(3*t1 + t0 + 8); + } + out[w*2-1] = stbi__div4(t1+2); + + STBI_NOTUSED(hs); + + return out; +} + +#if defined(STBI_SSE2) || defined(STBI_NEON) +static stbi_uc *stbi__resample_row_hv_2_simd(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs) +{ + // need to generate 2x2 samples for every one in input + int i=0,t0,t1; + + if (w == 1) { + out[0] = out[1] = stbi__div4(3*in_near[0] + in_far[0] + 2); + return out; + } + + t1 = 3*in_near[0] + in_far[0]; + // process groups of 8 pixels for as long as we can. + // note we can't handle the last pixel in a row in this loop + // because we need to handle the filter boundary conditions. + for (; i < ((w-1) & ~7); i += 8) { +#if defined(STBI_SSE2) + // load and perform the vertical filtering pass + // this uses 3*x + y = 4*x + (y - x) + __m128i zero = _mm_setzero_si128(); + __m128i farb = _mm_loadl_epi64((__m128i *) (in_far + i)); + __m128i nearb = _mm_loadl_epi64((__m128i *) (in_near + i)); + __m128i farw = _mm_unpacklo_epi8(farb, zero); + __m128i nearw = _mm_unpacklo_epi8(nearb, zero); + __m128i diff = _mm_sub_epi16(farw, nearw); + __m128i nears = _mm_slli_epi16(nearw, 2); + __m128i curr = _mm_add_epi16(nears, diff); // current row + + // horizontal filter works the same based on shifted vers of current + // row. "prev" is current row shifted right by 1 pixel; we need to + // insert the previous pixel value (from t1). + // "next" is current row shifted left by 1 pixel, with first pixel + // of next block of 8 pixels added in. + __m128i prv0 = _mm_slli_si128(curr, 2); + __m128i nxt0 = _mm_srli_si128(curr, 2); + __m128i prev = _mm_insert_epi16(prv0, t1, 0); + __m128i next = _mm_insert_epi16(nxt0, 3*in_near[i+8] + in_far[i+8], 7); + + // horizontal filter, polyphase implementation since it's convenient: + // even pixels = 3*cur + prev = cur*4 + (prev - cur) + // odd pixels = 3*cur + next = cur*4 + (next - cur) + // note the shared term. + __m128i bias = _mm_set1_epi16(8); + __m128i curs = _mm_slli_epi16(curr, 2); + __m128i prvd = _mm_sub_epi16(prev, curr); + __m128i nxtd = _mm_sub_epi16(next, curr); + __m128i curb = _mm_add_epi16(curs, bias); + __m128i even = _mm_add_epi16(prvd, curb); + __m128i odd = _mm_add_epi16(nxtd, curb); + + // interleave even and odd pixels, then undo scaling. + __m128i int0 = _mm_unpacklo_epi16(even, odd); + __m128i int1 = _mm_unpackhi_epi16(even, odd); + __m128i de0 = _mm_srli_epi16(int0, 4); + __m128i de1 = _mm_srli_epi16(int1, 4); + + // pack and write output + __m128i outv = _mm_packus_epi16(de0, de1); + _mm_storeu_si128((__m128i *) (out + i*2), outv); +#elif defined(STBI_NEON) + // load and perform the vertical filtering pass + // this uses 3*x + y = 4*x + (y - x) + uint8x8_t farb = vld1_u8(in_far + i); + uint8x8_t nearb = vld1_u8(in_near + i); + int16x8_t diff = vreinterpretq_s16_u16(vsubl_u8(farb, nearb)); + int16x8_t nears = vreinterpretq_s16_u16(vshll_n_u8(nearb, 2)); + int16x8_t curr = vaddq_s16(nears, diff); // current row + + // horizontal filter works the same based on shifted vers of current + // row. "prev" is current row shifted right by 1 pixel; we need to + // insert the previous pixel value (from t1). + // "next" is current row shifted left by 1 pixel, with first pixel + // of next block of 8 pixels added in. + int16x8_t prv0 = vextq_s16(curr, curr, 7); + int16x8_t nxt0 = vextq_s16(curr, curr, 1); + int16x8_t prev = vsetq_lane_s16(t1, prv0, 0); + int16x8_t next = vsetq_lane_s16(3*in_near[i+8] + in_far[i+8], nxt0, 7); + + // horizontal filter, polyphase implementation since it's convenient: + // even pixels = 3*cur + prev = cur*4 + (prev - cur) + // odd pixels = 3*cur + next = cur*4 + (next - cur) + // note the shared term. + int16x8_t curs = vshlq_n_s16(curr, 2); + int16x8_t prvd = vsubq_s16(prev, curr); + int16x8_t nxtd = vsubq_s16(next, curr); + int16x8_t even = vaddq_s16(curs, prvd); + int16x8_t odd = vaddq_s16(curs, nxtd); + + // undo scaling and round, then store with even/odd phases interleaved + uint8x8x2_t o; + o.val[0] = vqrshrun_n_s16(even, 4); + o.val[1] = vqrshrun_n_s16(odd, 4); + vst2_u8(out + i*2, o); +#endif + + // "previous" value for next iter + t1 = 3*in_near[i+7] + in_far[i+7]; + } + + t0 = t1; + t1 = 3*in_near[i] + in_far[i]; + out[i*2] = stbi__div16(3*t1 + t0 + 8); + + for (++i; i < w; ++i) { + t0 = t1; + t1 = 3*in_near[i]+in_far[i]; + out[i*2-1] = stbi__div16(3*t0 + t1 + 8); + out[i*2 ] = stbi__div16(3*t1 + t0 + 8); + } + out[w*2-1] = stbi__div4(t1+2); + + STBI_NOTUSED(hs); + + return out; +} +#endif + +static stbi_uc *stbi__resample_row_generic(stbi_uc *out, stbi_uc *in_near, stbi_uc *in_far, int w, int hs) +{ + // resample with nearest-neighbor + int i,j; + STBI_NOTUSED(in_far); + for (i=0; i < w; ++i) + for (j=0; j < hs; ++j) + out[i*hs+j] = in_near[i]; + return out; +} + +// this is a reduced-precision calculation of YCbCr-to-RGB introduced +// to make sure the code produces the same results in both SIMD and scalar +#define stbi__float2fixed(x) (((int) ((x) * 4096.0f + 0.5f)) << 8) +static void stbi__YCbCr_to_RGB_row(stbi_uc *out, const stbi_uc *y, const stbi_uc *pcb, const stbi_uc *pcr, int count, int step) +{ + int i; + for (i=0; i < count; ++i) { + int y_fixed = (y[i] << 20) + (1<<19); // rounding + int r,g,b; + int cr = pcr[i] - 128; + int cb = pcb[i] - 128; + r = y_fixed + cr* stbi__float2fixed(1.40200f); + g = y_fixed + (cr*-stbi__float2fixed(0.71414f)) + ((cb*-stbi__float2fixed(0.34414f)) & 0xffff0000); + b = y_fixed + cb* stbi__float2fixed(1.77200f); + r >>= 20; + g >>= 20; + b >>= 20; + if ((unsigned) r > 255) { if (r < 0) r = 0; else r = 255; } + if ((unsigned) g > 255) { if (g < 0) g = 0; else g = 255; } + if ((unsigned) b > 255) { if (b < 0) b = 0; else b = 255; } + out[0] = (stbi_uc)r; + out[1] = (stbi_uc)g; + out[2] = (stbi_uc)b; + out[3] = 255; + out += step; + } +} + +#if defined(STBI_SSE2) || defined(STBI_NEON) +static void stbi__YCbCr_to_RGB_simd(stbi_uc *out, stbi_uc const *y, stbi_uc const *pcb, stbi_uc const *pcr, int count, int step) +{ + int i = 0; + +#ifdef STBI_SSE2 + // step == 3 is pretty ugly on the final interleave, and i'm not convinced + // it's useful in practice (you wouldn't use it for textures, for example). + // so just accelerate step == 4 case. + if (step == 4) { + // this is a fairly straightforward implementation and not super-optimized. + __m128i signflip = _mm_set1_epi8(-0x80); + __m128i cr_const0 = _mm_set1_epi16( (short) ( 1.40200f*4096.0f+0.5f)); + __m128i cr_const1 = _mm_set1_epi16( - (short) ( 0.71414f*4096.0f+0.5f)); + __m128i cb_const0 = _mm_set1_epi16( - (short) ( 0.34414f*4096.0f+0.5f)); + __m128i cb_const1 = _mm_set1_epi16( (short) ( 1.77200f*4096.0f+0.5f)); + __m128i y_bias = _mm_set1_epi8((char) (unsigned char) 128); + __m128i xw = _mm_set1_epi16(255); // alpha channel + + for (; i+7 < count; i += 8) { + // load + __m128i y_bytes = _mm_loadl_epi64((__m128i *) (y+i)); + __m128i cr_bytes = _mm_loadl_epi64((__m128i *) (pcr+i)); + __m128i cb_bytes = _mm_loadl_epi64((__m128i *) (pcb+i)); + __m128i cr_biased = _mm_xor_si128(cr_bytes, signflip); // -128 + __m128i cb_biased = _mm_xor_si128(cb_bytes, signflip); // -128 + + // unpack to short (and left-shift cr, cb by 8) + __m128i yw = _mm_unpacklo_epi8(y_bias, y_bytes); + __m128i crw = _mm_unpacklo_epi8(_mm_setzero_si128(), cr_biased); + __m128i cbw = _mm_unpacklo_epi8(_mm_setzero_si128(), cb_biased); + + // color transform + __m128i yws = _mm_srli_epi16(yw, 4); + __m128i cr0 = _mm_mulhi_epi16(cr_const0, crw); + __m128i cb0 = _mm_mulhi_epi16(cb_const0, cbw); + __m128i cb1 = _mm_mulhi_epi16(cbw, cb_const1); + __m128i cr1 = _mm_mulhi_epi16(crw, cr_const1); + __m128i rws = _mm_add_epi16(cr0, yws); + __m128i gwt = _mm_add_epi16(cb0, yws); + __m128i bws = _mm_add_epi16(yws, cb1); + __m128i gws = _mm_add_epi16(gwt, cr1); + + // descale + __m128i rw = _mm_srai_epi16(rws, 4); + __m128i bw = _mm_srai_epi16(bws, 4); + __m128i gw = _mm_srai_epi16(gws, 4); + + // back to byte, set up for transpose + __m128i brb = _mm_packus_epi16(rw, bw); + __m128i gxb = _mm_packus_epi16(gw, xw); + + // transpose to interleave channels + __m128i t0 = _mm_unpacklo_epi8(brb, gxb); + __m128i t1 = _mm_unpackhi_epi8(brb, gxb); + __m128i o0 = _mm_unpacklo_epi16(t0, t1); + __m128i o1 = _mm_unpackhi_epi16(t0, t1); + + // store + _mm_storeu_si128((__m128i *) (out + 0), o0); + _mm_storeu_si128((__m128i *) (out + 16), o1); + out += 32; + } + } +#endif + +#ifdef STBI_NEON + // in this version, step=3 support would be easy to add. but is there demand? + if (step == 4) { + // this is a fairly straightforward implementation and not super-optimized. + uint8x8_t signflip = vdup_n_u8(0x80); + int16x8_t cr_const0 = vdupq_n_s16( (short) ( 1.40200f*4096.0f+0.5f)); + int16x8_t cr_const1 = vdupq_n_s16( - (short) ( 0.71414f*4096.0f+0.5f)); + int16x8_t cb_const0 = vdupq_n_s16( - (short) ( 0.34414f*4096.0f+0.5f)); + int16x8_t cb_const1 = vdupq_n_s16( (short) ( 1.77200f*4096.0f+0.5f)); + + for (; i+7 < count; i += 8) { + // load + uint8x8_t y_bytes = vld1_u8(y + i); + uint8x8_t cr_bytes = vld1_u8(pcr + i); + uint8x8_t cb_bytes = vld1_u8(pcb + i); + int8x8_t cr_biased = vreinterpret_s8_u8(vsub_u8(cr_bytes, signflip)); + int8x8_t cb_biased = vreinterpret_s8_u8(vsub_u8(cb_bytes, signflip)); + + // expand to s16 + int16x8_t yws = vreinterpretq_s16_u16(vshll_n_u8(y_bytes, 4)); + int16x8_t crw = vshll_n_s8(cr_biased, 7); + int16x8_t cbw = vshll_n_s8(cb_biased, 7); + + // color transform + int16x8_t cr0 = vqdmulhq_s16(crw, cr_const0); + int16x8_t cb0 = vqdmulhq_s16(cbw, cb_const0); + int16x8_t cr1 = vqdmulhq_s16(crw, cr_const1); + int16x8_t cb1 = vqdmulhq_s16(cbw, cb_const1); + int16x8_t rws = vaddq_s16(yws, cr0); + int16x8_t gws = vaddq_s16(vaddq_s16(yws, cb0), cr1); + int16x8_t bws = vaddq_s16(yws, cb1); + + // undo scaling, round, convert to byte + uint8x8x4_t o; + o.val[0] = vqrshrun_n_s16(rws, 4); + o.val[1] = vqrshrun_n_s16(gws, 4); + o.val[2] = vqrshrun_n_s16(bws, 4); + o.val[3] = vdup_n_u8(255); + + // store, interleaving r/g/b/a + vst4_u8(out, o); + out += 8*4; + } + } +#endif + + for (; i < count; ++i) { + int y_fixed = (y[i] << 20) + (1<<19); // rounding + int r,g,b; + int cr = pcr[i] - 128; + int cb = pcb[i] - 128; + r = y_fixed + cr* stbi__float2fixed(1.40200f); + g = y_fixed + cr*-stbi__float2fixed(0.71414f) + ((cb*-stbi__float2fixed(0.34414f)) & 0xffff0000); + b = y_fixed + cb* stbi__float2fixed(1.77200f); + r >>= 20; + g >>= 20; + b >>= 20; + if ((unsigned) r > 255) { if (r < 0) r = 0; else r = 255; } + if ((unsigned) g > 255) { if (g < 0) g = 0; else g = 255; } + if ((unsigned) b > 255) { if (b < 0) b = 0; else b = 255; } + out[0] = (stbi_uc)r; + out[1] = (stbi_uc)g; + out[2] = (stbi_uc)b; + out[3] = 255; + out += step; + } +} +#endif + +// set up the kernels +static void stbi__setup_jpeg(stbi__jpeg *j) +{ + j->idct_block_kernel = stbi__idct_block; + j->YCbCr_to_RGB_kernel = stbi__YCbCr_to_RGB_row; + j->resample_row_hv_2_kernel = stbi__resample_row_hv_2; + +#ifdef STBI_SSE2 + if (stbi__sse2_available()) { + j->idct_block_kernel = stbi__idct_simd; + j->YCbCr_to_RGB_kernel = stbi__YCbCr_to_RGB_simd; + j->resample_row_hv_2_kernel = stbi__resample_row_hv_2_simd; + } +#endif + +#ifdef STBI_NEON + j->idct_block_kernel = stbi__idct_simd; + j->YCbCr_to_RGB_kernel = stbi__YCbCr_to_RGB_simd; + j->resample_row_hv_2_kernel = stbi__resample_row_hv_2_simd; +#endif +} + +// clean up the temporary component buffers +static void stbi__cleanup_jpeg(stbi__jpeg *j) +{ + stbi__free_jpeg_components(j, j->s->img_n, 0); +} + +typedef struct +{ + resample_row_func resample; + stbi_uc *line0,*line1; + int hs,vs; // expansion factor in each axis + int w_lores; // horizontal pixels pre-expansion + int ystep; // how far through vertical expansion we are + int ypos; // which pre-expansion row we're on +} stbi__resample; + +// fast 0..255 * 0..255 => 0..255 rounded multiplication +static stbi_uc stbi__blinn_8x8(stbi_uc x, stbi_uc y) +{ + unsigned int t = x*y + 128; + return (stbi_uc) ((t + (t >>8)) >> 8); +} + +static stbi_uc *load_jpeg_image(stbi__jpeg *z, int *out_x, int *out_y, int *comp, int req_comp) +{ + int n, decode_n, is_rgb; + z->s->img_n = 0; // make stbi__cleanup_jpeg safe + + // validate req_comp + if (req_comp < 0 || req_comp > 4) return stbi__errpuc("bad req_comp", "Internal error"); + + // load a jpeg image from whichever source, but leave in YCbCr format + if (!stbi__decode_jpeg_image(z)) { stbi__cleanup_jpeg(z); return NULL; } + + // determine actual number of components to generate + n = req_comp ? req_comp : z->s->img_n >= 3 ? 3 : 1; + + is_rgb = z->s->img_n == 3 && (z->rgb == 3 || (z->app14_color_transform == 0 && !z->jfif)); + + if (z->s->img_n == 3 && n < 3 && !is_rgb) + decode_n = 1; + else + decode_n = z->s->img_n; + + // nothing to do if no components requested; check this now to avoid + // accessing uninitialized coutput[0] later + if (decode_n <= 0) { stbi__cleanup_jpeg(z); return NULL; } + + // resample and color-convert + { + int k; + unsigned int i,j; + stbi_uc *output; + stbi_uc *coutput[4] = { NULL, NULL, NULL, NULL }; + + stbi__resample res_comp[4]; + + for (k=0; k < decode_n; ++k) { + stbi__resample *r = &res_comp[k]; + + // allocate line buffer big enough for upsampling off the edges + // with upsample factor of 4 + z->img_comp[k].linebuf = (stbi_uc *) stbi__malloc(z->s->img_x + 3); + if (!z->img_comp[k].linebuf) { stbi__cleanup_jpeg(z); return stbi__errpuc("outofmem", "Out of memory"); } + + r->hs = z->img_h_max / z->img_comp[k].h; + r->vs = z->img_v_max / z->img_comp[k].v; + r->ystep = r->vs >> 1; + r->w_lores = (z->s->img_x + r->hs-1) / r->hs; + r->ypos = 0; + r->line0 = r->line1 = z->img_comp[k].data; + + if (r->hs == 1 && r->vs == 1) r->resample = resample_row_1; + else if (r->hs == 1 && r->vs == 2) r->resample = stbi__resample_row_v_2; + else if (r->hs == 2 && r->vs == 1) r->resample = stbi__resample_row_h_2; + else if (r->hs == 2 && r->vs == 2) r->resample = z->resample_row_hv_2_kernel; + else r->resample = stbi__resample_row_generic; + } + + // can't error after this so, this is safe + output = (stbi_uc *) stbi__malloc_mad3(n, z->s->img_x, z->s->img_y, 1); + if (!output) { stbi__cleanup_jpeg(z); return stbi__errpuc("outofmem", "Out of memory"); } + + // now go ahead and resample + for (j=0; j < z->s->img_y; ++j) { + stbi_uc *out = output + n * z->s->img_x * j; + for (k=0; k < decode_n; ++k) { + stbi__resample *r = &res_comp[k]; + int y_bot = r->ystep >= (r->vs >> 1); + coutput[k] = r->resample(z->img_comp[k].linebuf, + y_bot ? r->line1 : r->line0, + y_bot ? r->line0 : r->line1, + r->w_lores, r->hs); + if (++r->ystep >= r->vs) { + r->ystep = 0; + r->line0 = r->line1; + if (++r->ypos < z->img_comp[k].y) + r->line1 += z->img_comp[k].w2; + } + } + if (n >= 3) { + stbi_uc *y = coutput[0]; + if (z->s->img_n == 3) { + if (is_rgb) { + for (i=0; i < z->s->img_x; ++i) { + out[0] = y[i]; + out[1] = coutput[1][i]; + out[2] = coutput[2][i]; + out[3] = 255; + out += n; + } + } else { + z->YCbCr_to_RGB_kernel(out, y, coutput[1], coutput[2], z->s->img_x, n); + } + } else if (z->s->img_n == 4) { + if (z->app14_color_transform == 0) { // CMYK + for (i=0; i < z->s->img_x; ++i) { + stbi_uc m = coutput[3][i]; + out[0] = stbi__blinn_8x8(coutput[0][i], m); + out[1] = stbi__blinn_8x8(coutput[1][i], m); + out[2] = stbi__blinn_8x8(coutput[2][i], m); + out[3] = 255; + out += n; + } + } else if (z->app14_color_transform == 2) { // YCCK + z->YCbCr_to_RGB_kernel(out, y, coutput[1], coutput[2], z->s->img_x, n); + for (i=0; i < z->s->img_x; ++i) { + stbi_uc m = coutput[3][i]; + out[0] = stbi__blinn_8x8(255 - out[0], m); + out[1] = stbi__blinn_8x8(255 - out[1], m); + out[2] = stbi__blinn_8x8(255 - out[2], m); + out += n; + } + } else { // YCbCr + alpha? Ignore the fourth channel for now + z->YCbCr_to_RGB_kernel(out, y, coutput[1], coutput[2], z->s->img_x, n); + } + } else + for (i=0; i < z->s->img_x; ++i) { + out[0] = out[1] = out[2] = y[i]; + out[3] = 255; // not used if n==3 + out += n; + } + } else { + if (is_rgb) { + if (n == 1) + for (i=0; i < z->s->img_x; ++i) + *out++ = stbi__compute_y(coutput[0][i], coutput[1][i], coutput[2][i]); + else { + for (i=0; i < z->s->img_x; ++i, out += 2) { + out[0] = stbi__compute_y(coutput[0][i], coutput[1][i], coutput[2][i]); + out[1] = 255; + } + } + } else if (z->s->img_n == 4 && z->app14_color_transform == 0) { + for (i=0; i < z->s->img_x; ++i) { + stbi_uc m = coutput[3][i]; + stbi_uc r = stbi__blinn_8x8(coutput[0][i], m); + stbi_uc g = stbi__blinn_8x8(coutput[1][i], m); + stbi_uc b = stbi__blinn_8x8(coutput[2][i], m); + out[0] = stbi__compute_y(r, g, b); + out[1] = 255; + out += n; + } + } else if (z->s->img_n == 4 && z->app14_color_transform == 2) { + for (i=0; i < z->s->img_x; ++i) { + out[0] = stbi__blinn_8x8(255 - coutput[0][i], coutput[3][i]); + out[1] = 255; + out += n; + } + } else { + stbi_uc *y = coutput[0]; + if (n == 1) + for (i=0; i < z->s->img_x; ++i) out[i] = y[i]; + else + for (i=0; i < z->s->img_x; ++i) { *out++ = y[i]; *out++ = 255; } + } + } + } + stbi__cleanup_jpeg(z); + *out_x = z->s->img_x; + *out_y = z->s->img_y; + if (comp) *comp = z->s->img_n >= 3 ? 3 : 1; // report original components, not output + return output; + } +} + +static void *stbi__jpeg_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + unsigned char* result; + stbi__jpeg* j = (stbi__jpeg*) stbi__malloc(sizeof(stbi__jpeg)); + if (!j) return stbi__errpuc("outofmem", "Out of memory"); + memset(j, 0, sizeof(stbi__jpeg)); + STBI_NOTUSED(ri); + j->s = s; + stbi__setup_jpeg(j); + result = load_jpeg_image(j, x,y,comp,req_comp); + STBI_FREE(j); + return result; +} + +static int stbi__jpeg_test(stbi__context *s) +{ + int r; + stbi__jpeg* j = (stbi__jpeg*)stbi__malloc(sizeof(stbi__jpeg)); + if (!j) return stbi__err("outofmem", "Out of memory"); + memset(j, 0, sizeof(stbi__jpeg)); + j->s = s; + stbi__setup_jpeg(j); + r = stbi__decode_jpeg_header(j, STBI__SCAN_type); + stbi__rewind(s); + STBI_FREE(j); + return r; +} + +static int stbi__jpeg_info_raw(stbi__jpeg *j, int *x, int *y, int *comp) +{ + if (!stbi__decode_jpeg_header(j, STBI__SCAN_header)) { + stbi__rewind( j->s ); + return 0; + } + if (x) *x = j->s->img_x; + if (y) *y = j->s->img_y; + if (comp) *comp = j->s->img_n >= 3 ? 3 : 1; + return 1; +} + +static int stbi__jpeg_info(stbi__context *s, int *x, int *y, int *comp) +{ + int result; + stbi__jpeg* j = (stbi__jpeg*) (stbi__malloc(sizeof(stbi__jpeg))); + if (!j) return stbi__err("outofmem", "Out of memory"); + memset(j, 0, sizeof(stbi__jpeg)); + j->s = s; + result = stbi__jpeg_info_raw(j, x, y, comp); + STBI_FREE(j); + return result; +} +#endif + +// public domain zlib decode v0.2 Sean Barrett 2006-11-18 +// simple implementation +// - all input must be provided in an upfront buffer +// - all output is written to a single output buffer (can malloc/realloc) +// performance +// - fast huffman + +#ifndef STBI_NO_ZLIB + +// fast-way is faster to check than jpeg huffman, but slow way is slower +#define STBI__ZFAST_BITS 9 // accelerate all cases in default tables +#define STBI__ZFAST_MASK ((1 << STBI__ZFAST_BITS) - 1) +#define STBI__ZNSYMS 288 // number of symbols in literal/length alphabet + +// zlib-style huffman encoding +// (jpegs packs from left, zlib from right, so can't share code) +typedef struct +{ + stbi__uint16 fast[1 << STBI__ZFAST_BITS]; + stbi__uint16 firstcode[16]; + int maxcode[17]; + stbi__uint16 firstsymbol[16]; + stbi_uc size[STBI__ZNSYMS]; + stbi__uint16 value[STBI__ZNSYMS]; +} stbi__zhuffman; + +stbi_inline static int stbi__bitreverse16(int n) +{ + n = ((n & 0xAAAA) >> 1) | ((n & 0x5555) << 1); + n = ((n & 0xCCCC) >> 2) | ((n & 0x3333) << 2); + n = ((n & 0xF0F0) >> 4) | ((n & 0x0F0F) << 4); + n = ((n & 0xFF00) >> 8) | ((n & 0x00FF) << 8); + return n; +} + +stbi_inline static int stbi__bit_reverse(int v, int bits) +{ + STBI_ASSERT(bits <= 16); + // to bit reverse n bits, reverse 16 and shift + // e.g. 11 bits, bit reverse and shift away 5 + return stbi__bitreverse16(v) >> (16-bits); +} + +static int stbi__zbuild_huffman(stbi__zhuffman *z, const stbi_uc *sizelist, int num) +{ + int i,k=0; + int code, next_code[16], sizes[17]; + + // DEFLATE spec for generating codes + memset(sizes, 0, sizeof(sizes)); + memset(z->fast, 0, sizeof(z->fast)); + for (i=0; i < num; ++i) + ++sizes[sizelist[i]]; + sizes[0] = 0; + for (i=1; i < 16; ++i) + if (sizes[i] > (1 << i)) + return stbi__err("bad sizes", "Corrupt PNG"); + code = 0; + for (i=1; i < 16; ++i) { + next_code[i] = code; + z->firstcode[i] = (stbi__uint16) code; + z->firstsymbol[i] = (stbi__uint16) k; + code = (code + sizes[i]); + if (sizes[i]) + if (code-1 >= (1 << i)) return stbi__err("bad codelengths","Corrupt PNG"); + z->maxcode[i] = code << (16-i); // preshift for inner loop + code <<= 1; + k += sizes[i]; + } + z->maxcode[16] = 0x10000; // sentinel + for (i=0; i < num; ++i) { + int s = sizelist[i]; + if (s) { + int c = next_code[s] - z->firstcode[s] + z->firstsymbol[s]; + stbi__uint16 fastv = (stbi__uint16) ((s << 9) | i); + z->size [c] = (stbi_uc ) s; + z->value[c] = (stbi__uint16) i; + if (s <= STBI__ZFAST_BITS) { + int j = stbi__bit_reverse(next_code[s],s); + while (j < (1 << STBI__ZFAST_BITS)) { + z->fast[j] = fastv; + j += (1 << s); + } + } + ++next_code[s]; + } + } + return 1; +} + +// zlib-from-memory implementation for PNG reading +// because PNG allows splitting the zlib stream arbitrarily, +// and it's annoying structurally to have PNG call ZLIB call PNG, +// we require PNG read all the IDATs and combine them into a single +// memory buffer + +typedef struct +{ + stbi_uc *zbuffer, *zbuffer_end; + int num_bits; + int hit_zeof_once; + stbi__uint32 code_buffer; + + char *zout; + char *zout_start; + char *zout_end; + int z_expandable; + + stbi__zhuffman z_length, z_distance; +} stbi__zbuf; + +stbi_inline static int stbi__zeof(stbi__zbuf *z) +{ + return (z->zbuffer >= z->zbuffer_end); +} + +stbi_inline static stbi_uc stbi__zget8(stbi__zbuf *z) +{ + return stbi__zeof(z) ? 0 : *z->zbuffer++; +} + +static void stbi__fill_bits(stbi__zbuf *z) +{ + do { + if (z->code_buffer >= (1U << z->num_bits)) { + z->zbuffer = z->zbuffer_end; /* treat this as EOF so we fail. */ + return; + } + z->code_buffer |= (unsigned int) stbi__zget8(z) << z->num_bits; + z->num_bits += 8; + } while (z->num_bits <= 24); +} + +stbi_inline static unsigned int stbi__zreceive(stbi__zbuf *z, int n) +{ + unsigned int k; + if (z->num_bits < n) stbi__fill_bits(z); + k = z->code_buffer & ((1 << n) - 1); + z->code_buffer >>= n; + z->num_bits -= n; + return k; +} + +static int stbi__zhuffman_decode_slowpath(stbi__zbuf *a, stbi__zhuffman *z) +{ + int b,s,k; + // not resolved by fast table, so compute it the slow way + // use jpeg approach, which requires MSbits at top + k = stbi__bit_reverse(a->code_buffer, 16); + for (s=STBI__ZFAST_BITS+1; ; ++s) + if (k < z->maxcode[s]) + break; + if (s >= 16) return -1; // invalid code! + // code size is s, so: + b = (k >> (16-s)) - z->firstcode[s] + z->firstsymbol[s]; + if (b >= STBI__ZNSYMS) return -1; // some data was corrupt somewhere! + if (z->size[b] != s) return -1; // was originally an assert, but report failure instead. + a->code_buffer >>= s; + a->num_bits -= s; + return z->value[b]; +} + +stbi_inline static int stbi__zhuffman_decode(stbi__zbuf *a, stbi__zhuffman *z) +{ + int b,s; + if (a->num_bits < 16) { + if (stbi__zeof(a)) { + if (!a->hit_zeof_once) { + // This is the first time we hit eof, insert 16 extra padding btis + // to allow us to keep going; if we actually consume any of them + // though, that is invalid data. This is caught later. + a->hit_zeof_once = 1; + a->num_bits += 16; // add 16 implicit zero bits + } else { + // We already inserted our extra 16 padding bits and are again + // out, this stream is actually prematurely terminated. + return -1; + } + } else { + stbi__fill_bits(a); + } + } + b = z->fast[a->code_buffer & STBI__ZFAST_MASK]; + if (b) { + s = b >> 9; + a->code_buffer >>= s; + a->num_bits -= s; + return b & 511; + } + return stbi__zhuffman_decode_slowpath(a, z); +} + +static int stbi__zexpand(stbi__zbuf *z, char *zout, int n) // need to make room for n bytes +{ + char *q; + unsigned int cur, limit, old_limit; + z->zout = zout; + if (!z->z_expandable) return stbi__err("output buffer limit","Corrupt PNG"); + cur = (unsigned int) (z->zout - z->zout_start); + limit = old_limit = (unsigned) (z->zout_end - z->zout_start); + if (UINT_MAX - cur < (unsigned) n) return stbi__err("outofmem", "Out of memory"); + while (cur + n > limit) { + if(limit > UINT_MAX / 2) return stbi__err("outofmem", "Out of memory"); + limit *= 2; + } + q = (char *) STBI_REALLOC_SIZED(z->zout_start, old_limit, limit); + STBI_NOTUSED(old_limit); + if (q == NULL) return stbi__err("outofmem", "Out of memory"); + z->zout_start = q; + z->zout = q + cur; + z->zout_end = q + limit; + return 1; +} + +static const int stbi__zlength_base[31] = { + 3,4,5,6,7,8,9,10,11,13, + 15,17,19,23,27,31,35,43,51,59, + 67,83,99,115,131,163,195,227,258,0,0 }; + +static const int stbi__zlength_extra[31]= +{ 0,0,0,0,0,0,0,0,1,1,1,1,2,2,2,2,3,3,3,3,4,4,4,4,5,5,5,5,0,0,0 }; + +static const int stbi__zdist_base[32] = { 1,2,3,4,5,7,9,13,17,25,33,49,65,97,129,193, +257,385,513,769,1025,1537,2049,3073,4097,6145,8193,12289,16385,24577,0,0}; + +static const int stbi__zdist_extra[32] = +{ 0,0,0,0,1,1,2,2,3,3,4,4,5,5,6,6,7,7,8,8,9,9,10,10,11,11,12,12,13,13}; + +static int stbi__parse_huffman_block(stbi__zbuf *a) +{ + char *zout = a->zout; + for(;;) { + int z = stbi__zhuffman_decode(a, &a->z_length); + if (z < 256) { + if (z < 0) return stbi__err("bad huffman code","Corrupt PNG"); // error in huffman codes + if (zout >= a->zout_end) { + if (!stbi__zexpand(a, zout, 1)) return 0; + zout = a->zout; + } + *zout++ = (char) z; + } else { + stbi_uc *p; + int len,dist; + if (z == 256) { + a->zout = zout; + if (a->hit_zeof_once && a->num_bits < 16) { + // The first time we hit zeof, we inserted 16 extra zero bits into our bit + // buffer so the decoder can just do its speculative decoding. But if we + // actually consumed any of those bits (which is the case when num_bits < 16), + // the stream actually read past the end so it is malformed. + return stbi__err("unexpected end","Corrupt PNG"); + } + return 1; + } + if (z >= 286) return stbi__err("bad huffman code","Corrupt PNG"); // per DEFLATE, length codes 286 and 287 must not appear in compressed data + z -= 257; + len = stbi__zlength_base[z]; + if (stbi__zlength_extra[z]) len += stbi__zreceive(a, stbi__zlength_extra[z]); + z = stbi__zhuffman_decode(a, &a->z_distance); + if (z < 0 || z >= 30) return stbi__err("bad huffman code","Corrupt PNG"); // per DEFLATE, distance codes 30 and 31 must not appear in compressed data + dist = stbi__zdist_base[z]; + if (stbi__zdist_extra[z]) dist += stbi__zreceive(a, stbi__zdist_extra[z]); + if (zout - a->zout_start < dist) return stbi__err("bad dist","Corrupt PNG"); + if (len > a->zout_end - zout) { + if (!stbi__zexpand(a, zout, len)) return 0; + zout = a->zout; + } + p = (stbi_uc *) (zout - dist); + if (dist == 1) { // run of one byte; common in images. + stbi_uc v = *p; + if (len) { do *zout++ = v; while (--len); } + } else { + if (len) { do *zout++ = *p++; while (--len); } + } + } + } +} + +static int stbi__compute_huffman_codes(stbi__zbuf *a) +{ + static const stbi_uc length_dezigzag[19] = { 16,17,18,0,8,7,9,6,10,5,11,4,12,3,13,2,14,1,15 }; + stbi__zhuffman z_codelength; + stbi_uc lencodes[286+32+137];//padding for maximum single op + stbi_uc codelength_sizes[19]; + int i,n; + + int hlit = stbi__zreceive(a,5) + 257; + int hdist = stbi__zreceive(a,5) + 1; + int hclen = stbi__zreceive(a,4) + 4; + int ntot = hlit + hdist; + + memset(codelength_sizes, 0, sizeof(codelength_sizes)); + for (i=0; i < hclen; ++i) { + int s = stbi__zreceive(a,3); + codelength_sizes[length_dezigzag[i]] = (stbi_uc) s; + } + if (!stbi__zbuild_huffman(&z_codelength, codelength_sizes, 19)) return 0; + + n = 0; + while (n < ntot) { + int c = stbi__zhuffman_decode(a, &z_codelength); + if (c < 0 || c >= 19) return stbi__err("bad codelengths", "Corrupt PNG"); + if (c < 16) + lencodes[n++] = (stbi_uc) c; + else { + stbi_uc fill = 0; + if (c == 16) { + c = stbi__zreceive(a,2)+3; + if (n == 0) return stbi__err("bad codelengths", "Corrupt PNG"); + fill = lencodes[n-1]; + } else if (c == 17) { + c = stbi__zreceive(a,3)+3; + } else if (c == 18) { + c = stbi__zreceive(a,7)+11; + } else { + return stbi__err("bad codelengths", "Corrupt PNG"); + } + if (ntot - n < c) return stbi__err("bad codelengths", "Corrupt PNG"); + memset(lencodes+n, fill, c); + n += c; + } + } + if (n != ntot) return stbi__err("bad codelengths","Corrupt PNG"); + if (!stbi__zbuild_huffman(&a->z_length, lencodes, hlit)) return 0; + if (!stbi__zbuild_huffman(&a->z_distance, lencodes+hlit, hdist)) return 0; + return 1; +} + +static int stbi__parse_uncompressed_block(stbi__zbuf *a) +{ + stbi_uc header[4]; + int len,nlen,k; + if (a->num_bits & 7) + stbi__zreceive(a, a->num_bits & 7); // discard + // drain the bit-packed data into header + k = 0; + while (a->num_bits > 0) { + header[k++] = (stbi_uc) (a->code_buffer & 255); // suppress MSVC run-time check + a->code_buffer >>= 8; + a->num_bits -= 8; + } + if (a->num_bits < 0) return stbi__err("zlib corrupt","Corrupt PNG"); + // now fill header the normal way + while (k < 4) + header[k++] = stbi__zget8(a); + len = header[1] * 256 + header[0]; + nlen = header[3] * 256 + header[2]; + if (nlen != (len ^ 0xffff)) return stbi__err("zlib corrupt","Corrupt PNG"); + if (a->zbuffer + len > a->zbuffer_end) return stbi__err("read past buffer","Corrupt PNG"); + if (a->zout + len > a->zout_end) + if (!stbi__zexpand(a, a->zout, len)) return 0; + memcpy(a->zout, a->zbuffer, len); + a->zbuffer += len; + a->zout += len; + return 1; +} + +static int stbi__parse_zlib_header(stbi__zbuf *a) +{ + int cmf = stbi__zget8(a); + int cm = cmf & 15; + /* int cinfo = cmf >> 4; */ + int flg = stbi__zget8(a); + if (stbi__zeof(a)) return stbi__err("bad zlib header","Corrupt PNG"); // zlib spec + if ((cmf*256+flg) % 31 != 0) return stbi__err("bad zlib header","Corrupt PNG"); // zlib spec + if (flg & 32) return stbi__err("no preset dict","Corrupt PNG"); // preset dictionary not allowed in png + if (cm != 8) return stbi__err("bad compression","Corrupt PNG"); // DEFLATE required for png + // window = 1 << (8 + cinfo)... but who cares, we fully buffer output + return 1; +} + +static const stbi_uc stbi__zdefault_length[STBI__ZNSYMS] = +{ + 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, + 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, + 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, + 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, + 8,8,8,8,8,8,8,8,8,8,8,8,8,8,8,8, 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, + 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, + 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, + 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, 9,9,9,9,9,9,9,9,9,9,9,9,9,9,9,9, + 7,7,7,7,7,7,7,7,7,7,7,7,7,7,7,7, 7,7,7,7,7,7,7,7,8,8,8,8,8,8,8,8 +}; +static const stbi_uc stbi__zdefault_distance[32] = +{ + 5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5,5 +}; +/* +Init algorithm: +{ + int i; // use <= to match clearly with spec + for (i=0; i <= 143; ++i) stbi__zdefault_length[i] = 8; + for ( ; i <= 255; ++i) stbi__zdefault_length[i] = 9; + for ( ; i <= 279; ++i) stbi__zdefault_length[i] = 7; + for ( ; i <= 287; ++i) stbi__zdefault_length[i] = 8; + + for (i=0; i <= 31; ++i) stbi__zdefault_distance[i] = 5; +} +*/ + +static int stbi__parse_zlib(stbi__zbuf *a, int parse_header) +{ + int final, type; + if (parse_header) + if (!stbi__parse_zlib_header(a)) return 0; + a->num_bits = 0; + a->code_buffer = 0; + a->hit_zeof_once = 0; + do { + final = stbi__zreceive(a,1); + type = stbi__zreceive(a,2); + if (type == 0) { + if (!stbi__parse_uncompressed_block(a)) return 0; + } else if (type == 3) { + return 0; + } else { + if (type == 1) { + // use fixed code lengths + if (!stbi__zbuild_huffman(&a->z_length , stbi__zdefault_length , STBI__ZNSYMS)) return 0; + if (!stbi__zbuild_huffman(&a->z_distance, stbi__zdefault_distance, 32)) return 0; + } else { + if (!stbi__compute_huffman_codes(a)) return 0; + } + if (!stbi__parse_huffman_block(a)) return 0; + } + } while (!final); + return 1; +} + +static int stbi__do_zlib(stbi__zbuf *a, char *obuf, int olen, int exp, int parse_header) +{ + a->zout_start = obuf; + a->zout = obuf; + a->zout_end = obuf + olen; + a->z_expandable = exp; + + return stbi__parse_zlib(a, parse_header); +} + +STBIDEF char *stbi_zlib_decode_malloc_guesssize(const char *buffer, int len, int initial_size, int *outlen) +{ + stbi__zbuf a; + char *p = (char *) stbi__malloc(initial_size); + if (p == NULL) return NULL; + a.zbuffer = (stbi_uc *) buffer; + a.zbuffer_end = (stbi_uc *) buffer + len; + if (stbi__do_zlib(&a, p, initial_size, 1, 1)) { + if (outlen) *outlen = (int) (a.zout - a.zout_start); + return a.zout_start; + } else { + STBI_FREE(a.zout_start); + return NULL; + } +} + +STBIDEF char *stbi_zlib_decode_malloc(char const *buffer, int len, int *outlen) +{ + return stbi_zlib_decode_malloc_guesssize(buffer, len, 16384, outlen); +} + +STBIDEF char *stbi_zlib_decode_malloc_guesssize_headerflag(const char *buffer, int len, int initial_size, int *outlen, int parse_header) +{ + stbi__zbuf a; + char *p = (char *) stbi__malloc(initial_size); + if (p == NULL) return NULL; + a.zbuffer = (stbi_uc *) buffer; + a.zbuffer_end = (stbi_uc *) buffer + len; + if (stbi__do_zlib(&a, p, initial_size, 1, parse_header)) { + if (outlen) *outlen = (int) (a.zout - a.zout_start); + return a.zout_start; + } else { + STBI_FREE(a.zout_start); + return NULL; + } +} + +STBIDEF int stbi_zlib_decode_buffer(char *obuffer, int olen, char const *ibuffer, int ilen) +{ + stbi__zbuf a; + a.zbuffer = (stbi_uc *) ibuffer; + a.zbuffer_end = (stbi_uc *) ibuffer + ilen; + if (stbi__do_zlib(&a, obuffer, olen, 0, 1)) + return (int) (a.zout - a.zout_start); + else + return -1; +} + +STBIDEF char *stbi_zlib_decode_noheader_malloc(char const *buffer, int len, int *outlen) +{ + stbi__zbuf a; + char *p = (char *) stbi__malloc(16384); + if (p == NULL) return NULL; + a.zbuffer = (stbi_uc *) buffer; + a.zbuffer_end = (stbi_uc *) buffer+len; + if (stbi__do_zlib(&a, p, 16384, 1, 0)) { + if (outlen) *outlen = (int) (a.zout - a.zout_start); + return a.zout_start; + } else { + STBI_FREE(a.zout_start); + return NULL; + } +} + +STBIDEF int stbi_zlib_decode_noheader_buffer(char *obuffer, int olen, const char *ibuffer, int ilen) +{ + stbi__zbuf a; + a.zbuffer = (stbi_uc *) ibuffer; + a.zbuffer_end = (stbi_uc *) ibuffer + ilen; + if (stbi__do_zlib(&a, obuffer, olen, 0, 0)) + return (int) (a.zout - a.zout_start); + else + return -1; +} +#endif + +// public domain "baseline" PNG decoder v0.10 Sean Barrett 2006-11-18 +// simple implementation +// - only 8-bit samples +// - no CRC checking +// - allocates lots of intermediate memory +// - avoids problem of streaming data between subsystems +// - avoids explicit window management +// performance +// - uses stb_zlib, a PD zlib implementation with fast huffman decoding + +#ifndef STBI_NO_PNG +typedef struct +{ + stbi__uint32 length; + stbi__uint32 type; +} stbi__pngchunk; + +static stbi__pngchunk stbi__get_chunk_header(stbi__context *s) +{ + stbi__pngchunk c; + c.length = stbi__get32be(s); + c.type = stbi__get32be(s); + return c; +} + +static int stbi__check_png_header(stbi__context *s) +{ + static const stbi_uc png_sig[8] = { 137,80,78,71,13,10,26,10 }; + int i; + for (i=0; i < 8; ++i) + if (stbi__get8(s) != png_sig[i]) return stbi__err("bad png sig","Not a PNG"); + return 1; +} + +typedef struct +{ + stbi__context *s; + stbi_uc *idata, *expanded, *out; + int depth; +} stbi__png; + + +enum { + STBI__F_none=0, + STBI__F_sub=1, + STBI__F_up=2, + STBI__F_avg=3, + STBI__F_paeth=4, + // synthetic filter used for first scanline to avoid needing a dummy row of 0s + STBI__F_avg_first +}; + +static stbi_uc first_row_filter[5] = +{ + STBI__F_none, + STBI__F_sub, + STBI__F_none, + STBI__F_avg_first, + STBI__F_sub // Paeth with b=c=0 turns out to be equivalent to sub +}; + +static int stbi__paeth(int a, int b, int c) +{ + // This formulation looks very different from the reference in the PNG spec, but is + // actually equivalent and has favorable data dependencies and admits straightforward + // generation of branch-free code, which helps performance significantly. + int thresh = c*3 - (a + b); + int lo = a < b ? a : b; + int hi = a < b ? b : a; + int t0 = (hi <= thresh) ? lo : c; + int t1 = (thresh <= lo) ? hi : t0; + return t1; +} + +static const stbi_uc stbi__depth_scale_table[9] = { 0, 0xff, 0x55, 0, 0x11, 0,0,0, 0x01 }; + +// adds an extra all-255 alpha channel +// dest == src is legal +// img_n must be 1 or 3 +static void stbi__create_png_alpha_expand8(stbi_uc *dest, stbi_uc *src, stbi__uint32 x, int img_n) +{ + int i; + // must process data backwards since we allow dest==src + if (img_n == 1) { + for (i=x-1; i >= 0; --i) { + dest[i*2+1] = 255; + dest[i*2+0] = src[i]; + } + } else { + STBI_ASSERT(img_n == 3); + for (i=x-1; i >= 0; --i) { + dest[i*4+3] = 255; + dest[i*4+2] = src[i*3+2]; + dest[i*4+1] = src[i*3+1]; + dest[i*4+0] = src[i*3+0]; + } + } +} + +// create the png data from post-deflated data +static int stbi__create_png_image_raw(stbi__png *a, stbi_uc *raw, stbi__uint32 raw_len, int out_n, stbi__uint32 x, stbi__uint32 y, int depth, int color) +{ + int bytes = (depth == 16 ? 2 : 1); + stbi__context *s = a->s; + stbi__uint32 i,j,stride = x*out_n*bytes; + stbi__uint32 img_len, img_width_bytes; + stbi_uc *filter_buf; + int all_ok = 1; + int k; + int img_n = s->img_n; // copy it into a local for later + + int output_bytes = out_n*bytes; + int filter_bytes = img_n*bytes; + int width = x; + + STBI_ASSERT(out_n == s->img_n || out_n == s->img_n+1); + a->out = (stbi_uc *) stbi__malloc_mad3(x, y, output_bytes, 0); // extra bytes to write off the end into + if (!a->out) return stbi__err("outofmem", "Out of memory"); + + // note: error exits here don't need to clean up a->out individually, + // stbi__do_png always does on error. + if (!stbi__mad3sizes_valid(img_n, x, depth, 7)) return stbi__err("too large", "Corrupt PNG"); + img_width_bytes = (((img_n * x * depth) + 7) >> 3); + if (!stbi__mad2sizes_valid(img_width_bytes, y, img_width_bytes)) return stbi__err("too large", "Corrupt PNG"); + img_len = (img_width_bytes + 1) * y; + + // we used to check for exact match between raw_len and img_len on non-interlaced PNGs, + // but issue #276 reported a PNG in the wild that had extra data at the end (all zeros), + // so just check for raw_len < img_len always. + if (raw_len < img_len) return stbi__err("not enough pixels","Corrupt PNG"); + + // Allocate two scan lines worth of filter workspace buffer. + filter_buf = (stbi_uc *) stbi__malloc_mad2(img_width_bytes, 2, 0); + if (!filter_buf) return stbi__err("outofmem", "Out of memory"); + + // Filtering for low-bit-depth images + if (depth < 8) { + filter_bytes = 1; + width = img_width_bytes; + } + + for (j=0; j < y; ++j) { + // cur/prior filter buffers alternate + stbi_uc *cur = filter_buf + (j & 1)*img_width_bytes; + stbi_uc *prior = filter_buf + (~j & 1)*img_width_bytes; + stbi_uc *dest = a->out + stride*j; + int nk = width * filter_bytes; + int filter = *raw++; + + // check filter type + if (filter > 4) { + all_ok = stbi__err("invalid filter","Corrupt PNG"); + break; + } + + // if first row, use special filter that doesn't sample previous row + if (j == 0) filter = first_row_filter[filter]; + + // perform actual filtering + switch (filter) { + case STBI__F_none: + memcpy(cur, raw, nk); + break; + case STBI__F_sub: + memcpy(cur, raw, filter_bytes); + for (k = filter_bytes; k < nk; ++k) + cur[k] = STBI__BYTECAST(raw[k] + cur[k-filter_bytes]); + break; + case STBI__F_up: + for (k = 0; k < nk; ++k) + cur[k] = STBI__BYTECAST(raw[k] + prior[k]); + break; + case STBI__F_avg: + for (k = 0; k < filter_bytes; ++k) + cur[k] = STBI__BYTECAST(raw[k] + (prior[k]>>1)); + for (k = filter_bytes; k < nk; ++k) + cur[k] = STBI__BYTECAST(raw[k] + ((prior[k] + cur[k-filter_bytes])>>1)); + break; + case STBI__F_paeth: + for (k = 0; k < filter_bytes; ++k) + cur[k] = STBI__BYTECAST(raw[k] + prior[k]); // prior[k] == stbi__paeth(0,prior[k],0) + for (k = filter_bytes; k < nk; ++k) + cur[k] = STBI__BYTECAST(raw[k] + stbi__paeth(cur[k-filter_bytes], prior[k], prior[k-filter_bytes])); + break; + case STBI__F_avg_first: + memcpy(cur, raw, filter_bytes); + for (k = filter_bytes; k < nk; ++k) + cur[k] = STBI__BYTECAST(raw[k] + (cur[k-filter_bytes] >> 1)); + break; + } + + raw += nk; + + // expand decoded bits in cur to dest, also adding an extra alpha channel if desired + if (depth < 8) { + stbi_uc scale = (color == 0) ? stbi__depth_scale_table[depth] : 1; // scale grayscale values to 0..255 range + stbi_uc *in = cur; + stbi_uc *out = dest; + stbi_uc inb = 0; + stbi__uint32 nsmp = x*img_n; + + // expand bits to bytes first + if (depth == 4) { + for (i=0; i < nsmp; ++i) { + if ((i & 1) == 0) inb = *in++; + *out++ = scale * (inb >> 4); + inb <<= 4; + } + } else if (depth == 2) { + for (i=0; i < nsmp; ++i) { + if ((i & 3) == 0) inb = *in++; + *out++ = scale * (inb >> 6); + inb <<= 2; + } + } else { + STBI_ASSERT(depth == 1); + for (i=0; i < nsmp; ++i) { + if ((i & 7) == 0) inb = *in++; + *out++ = scale * (inb >> 7); + inb <<= 1; + } + } + + // insert alpha=255 values if desired + if (img_n != out_n) + stbi__create_png_alpha_expand8(dest, dest, x, img_n); + } else if (depth == 8) { + if (img_n == out_n) + memcpy(dest, cur, x*img_n); + else + stbi__create_png_alpha_expand8(dest, cur, x, img_n); + } else if (depth == 16) { + // convert the image data from big-endian to platform-native + stbi__uint16 *dest16 = (stbi__uint16*)dest; + stbi__uint32 nsmp = x*img_n; + + if (img_n == out_n) { + for (i = 0; i < nsmp; ++i, ++dest16, cur += 2) + *dest16 = (cur[0] << 8) | cur[1]; + } else { + STBI_ASSERT(img_n+1 == out_n); + if (img_n == 1) { + for (i = 0; i < x; ++i, dest16 += 2, cur += 2) { + dest16[0] = (cur[0] << 8) | cur[1]; + dest16[1] = 0xffff; + } + } else { + STBI_ASSERT(img_n == 3); + for (i = 0; i < x; ++i, dest16 += 4, cur += 6) { + dest16[0] = (cur[0] << 8) | cur[1]; + dest16[1] = (cur[2] << 8) | cur[3]; + dest16[2] = (cur[4] << 8) | cur[5]; + dest16[3] = 0xffff; + } + } + } + } + } + + STBI_FREE(filter_buf); + if (!all_ok) return 0; + + return 1; +} + +static int stbi__create_png_image(stbi__png *a, stbi_uc *image_data, stbi__uint32 image_data_len, int out_n, int depth, int color, int interlaced) +{ + int bytes = (depth == 16 ? 2 : 1); + int out_bytes = out_n * bytes; + stbi_uc *final; + int p; + if (!interlaced) + return stbi__create_png_image_raw(a, image_data, image_data_len, out_n, a->s->img_x, a->s->img_y, depth, color); + + // de-interlacing + final = (stbi_uc *) stbi__malloc_mad3(a->s->img_x, a->s->img_y, out_bytes, 0); + if (!final) return stbi__err("outofmem", "Out of memory"); + for (p=0; p < 7; ++p) { + int xorig[] = { 0,4,0,2,0,1,0 }; + int yorig[] = { 0,0,4,0,2,0,1 }; + int xspc[] = { 8,8,4,4,2,2,1 }; + int yspc[] = { 8,8,8,4,4,2,2 }; + int i,j,x,y; + // pass1_x[4] = 0, pass1_x[5] = 1, pass1_x[12] = 1 + x = (a->s->img_x - xorig[p] + xspc[p]-1) / xspc[p]; + y = (a->s->img_y - yorig[p] + yspc[p]-1) / yspc[p]; + if (x && y) { + stbi__uint32 img_len = ((((a->s->img_n * x * depth) + 7) >> 3) + 1) * y; + if (!stbi__create_png_image_raw(a, image_data, image_data_len, out_n, x, y, depth, color)) { + STBI_FREE(final); + return 0; + } + for (j=0; j < y; ++j) { + for (i=0; i < x; ++i) { + int out_y = j*yspc[p]+yorig[p]; + int out_x = i*xspc[p]+xorig[p]; + memcpy(final + out_y*a->s->img_x*out_bytes + out_x*out_bytes, + a->out + (j*x+i)*out_bytes, out_bytes); + } + } + STBI_FREE(a->out); + image_data += img_len; + image_data_len -= img_len; + } + } + a->out = final; + + return 1; +} + +static int stbi__compute_transparency(stbi__png *z, stbi_uc tc[3], int out_n) +{ + stbi__context *s = z->s; + stbi__uint32 i, pixel_count = s->img_x * s->img_y; + stbi_uc *p = z->out; + + // compute color-based transparency, assuming we've + // already got 255 as the alpha value in the output + STBI_ASSERT(out_n == 2 || out_n == 4); + + if (out_n == 2) { + for (i=0; i < pixel_count; ++i) { + p[1] = (p[0] == tc[0] ? 0 : 255); + p += 2; + } + } else { + for (i=0; i < pixel_count; ++i) { + if (p[0] == tc[0] && p[1] == tc[1] && p[2] == tc[2]) + p[3] = 0; + p += 4; + } + } + return 1; +} + +static int stbi__compute_transparency16(stbi__png *z, stbi__uint16 tc[3], int out_n) +{ + stbi__context *s = z->s; + stbi__uint32 i, pixel_count = s->img_x * s->img_y; + stbi__uint16 *p = (stbi__uint16*) z->out; + + // compute color-based transparency, assuming we've + // already got 65535 as the alpha value in the output + STBI_ASSERT(out_n == 2 || out_n == 4); + + if (out_n == 2) { + for (i = 0; i < pixel_count; ++i) { + p[1] = (p[0] == tc[0] ? 0 : 65535); + p += 2; + } + } else { + for (i = 0; i < pixel_count; ++i) { + if (p[0] == tc[0] && p[1] == tc[1] && p[2] == tc[2]) + p[3] = 0; + p += 4; + } + } + return 1; +} + +static int stbi__expand_png_palette(stbi__png *a, stbi_uc *palette, int len, int pal_img_n) +{ + stbi__uint32 i, pixel_count = a->s->img_x * a->s->img_y; + stbi_uc *p, *temp_out, *orig = a->out; + + p = (stbi_uc *) stbi__malloc_mad2(pixel_count, pal_img_n, 0); + if (p == NULL) return stbi__err("outofmem", "Out of memory"); + + // between here and free(out) below, exitting would leak + temp_out = p; + + if (pal_img_n == 3) { + for (i=0; i < pixel_count; ++i) { + int n = orig[i]*4; + p[0] = palette[n ]; + p[1] = palette[n+1]; + p[2] = palette[n+2]; + p += 3; + } + } else { + for (i=0; i < pixel_count; ++i) { + int n = orig[i]*4; + p[0] = palette[n ]; + p[1] = palette[n+1]; + p[2] = palette[n+2]; + p[3] = palette[n+3]; + p += 4; + } + } + STBI_FREE(a->out); + a->out = temp_out; + + STBI_NOTUSED(len); + + return 1; +} + +static int stbi__unpremultiply_on_load_global = 0; +static int stbi__de_iphone_flag_global = 0; + +STBIDEF void stbi_set_unpremultiply_on_load(int flag_true_if_should_unpremultiply) +{ + stbi__unpremultiply_on_load_global = flag_true_if_should_unpremultiply; +} + +STBIDEF void stbi_convert_iphone_png_to_rgb(int flag_true_if_should_convert) +{ + stbi__de_iphone_flag_global = flag_true_if_should_convert; +} + +#ifndef STBI_THREAD_LOCAL +#define stbi__unpremultiply_on_load stbi__unpremultiply_on_load_global +#define stbi__de_iphone_flag stbi__de_iphone_flag_global +#else +static STBI_THREAD_LOCAL int stbi__unpremultiply_on_load_local, stbi__unpremultiply_on_load_set; +static STBI_THREAD_LOCAL int stbi__de_iphone_flag_local, stbi__de_iphone_flag_set; + +STBIDEF void stbi_set_unpremultiply_on_load_thread(int flag_true_if_should_unpremultiply) +{ + stbi__unpremultiply_on_load_local = flag_true_if_should_unpremultiply; + stbi__unpremultiply_on_load_set = 1; +} + +STBIDEF void stbi_convert_iphone_png_to_rgb_thread(int flag_true_if_should_convert) +{ + stbi__de_iphone_flag_local = flag_true_if_should_convert; + stbi__de_iphone_flag_set = 1; +} + +#define stbi__unpremultiply_on_load (stbi__unpremultiply_on_load_set \ + ? stbi__unpremultiply_on_load_local \ + : stbi__unpremultiply_on_load_global) +#define stbi__de_iphone_flag (stbi__de_iphone_flag_set \ + ? stbi__de_iphone_flag_local \ + : stbi__de_iphone_flag_global) +#endif // STBI_THREAD_LOCAL + +static void stbi__de_iphone(stbi__png *z) +{ + stbi__context *s = z->s; + stbi__uint32 i, pixel_count = s->img_x * s->img_y; + stbi_uc *p = z->out; + + if (s->img_out_n == 3) { // convert bgr to rgb + for (i=0; i < pixel_count; ++i) { + stbi_uc t = p[0]; + p[0] = p[2]; + p[2] = t; + p += 3; + } + } else { + STBI_ASSERT(s->img_out_n == 4); + if (stbi__unpremultiply_on_load) { + // convert bgr to rgb and unpremultiply + for (i=0; i < pixel_count; ++i) { + stbi_uc a = p[3]; + stbi_uc t = p[0]; + if (a) { + stbi_uc half = a / 2; + p[0] = (p[2] * 255 + half) / a; + p[1] = (p[1] * 255 + half) / a; + p[2] = ( t * 255 + half) / a; + } else { + p[0] = p[2]; + p[2] = t; + } + p += 4; + } + } else { + // convert bgr to rgb + for (i=0; i < pixel_count; ++i) { + stbi_uc t = p[0]; + p[0] = p[2]; + p[2] = t; + p += 4; + } + } + } +} + +#define STBI__PNG_TYPE(a,b,c,d) (((unsigned) (a) << 24) + ((unsigned) (b) << 16) + ((unsigned) (c) << 8) + (unsigned) (d)) + +static int stbi__parse_png_file(stbi__png *z, int scan, int req_comp) +{ + stbi_uc palette[1024], pal_img_n=0; + stbi_uc has_trans=0, tc[3]={0}; + stbi__uint16 tc16[3]; + stbi__uint32 ioff=0, idata_limit=0, i, pal_len=0; + int first=1,k,interlace=0, color=0, is_iphone=0; + stbi__context *s = z->s; + + z->expanded = NULL; + z->idata = NULL; + z->out = NULL; + + if (!stbi__check_png_header(s)) return 0; + + if (scan == STBI__SCAN_type) return 1; + + for (;;) { + stbi__pngchunk c = stbi__get_chunk_header(s); + switch (c.type) { + case STBI__PNG_TYPE('C','g','B','I'): + is_iphone = 1; + stbi__skip(s, c.length); + break; + case STBI__PNG_TYPE('I','H','D','R'): { + int comp,filter; + if (!first) return stbi__err("multiple IHDR","Corrupt PNG"); + first = 0; + if (c.length != 13) return stbi__err("bad IHDR len","Corrupt PNG"); + s->img_x = stbi__get32be(s); + s->img_y = stbi__get32be(s); + if (s->img_y > STBI_MAX_DIMENSIONS) return stbi__err("too large","Very large image (corrupt?)"); + if (s->img_x > STBI_MAX_DIMENSIONS) return stbi__err("too large","Very large image (corrupt?)"); + z->depth = stbi__get8(s); if (z->depth != 1 && z->depth != 2 && z->depth != 4 && z->depth != 8 && z->depth != 16) return stbi__err("1/2/4/8/16-bit only","PNG not supported: 1/2/4/8/16-bit only"); + color = stbi__get8(s); if (color > 6) return stbi__err("bad ctype","Corrupt PNG"); + if (color == 3 && z->depth == 16) return stbi__err("bad ctype","Corrupt PNG"); + if (color == 3) pal_img_n = 3; else if (color & 1) return stbi__err("bad ctype","Corrupt PNG"); + comp = stbi__get8(s); if (comp) return stbi__err("bad comp method","Corrupt PNG"); + filter= stbi__get8(s); if (filter) return stbi__err("bad filter method","Corrupt PNG"); + interlace = stbi__get8(s); if (interlace>1) return stbi__err("bad interlace method","Corrupt PNG"); + if (!s->img_x || !s->img_y) return stbi__err("0-pixel image","Corrupt PNG"); + if (!pal_img_n) { + s->img_n = (color & 2 ? 3 : 1) + (color & 4 ? 1 : 0); + if ((1 << 30) / s->img_x / s->img_n < s->img_y) return stbi__err("too large", "Image too large to decode"); + } else { + // if paletted, then pal_n is our final components, and + // img_n is # components to decompress/filter. + s->img_n = 1; + if ((1 << 30) / s->img_x / 4 < s->img_y) return stbi__err("too large","Corrupt PNG"); + } + // even with SCAN_header, have to scan to see if we have a tRNS + break; + } + + case STBI__PNG_TYPE('P','L','T','E'): { + if (first) return stbi__err("first not IHDR", "Corrupt PNG"); + if (c.length > 256*3) return stbi__err("invalid PLTE","Corrupt PNG"); + pal_len = c.length / 3; + if (pal_len * 3 != c.length) return stbi__err("invalid PLTE","Corrupt PNG"); + for (i=0; i < pal_len; ++i) { + palette[i*4+0] = stbi__get8(s); + palette[i*4+1] = stbi__get8(s); + palette[i*4+2] = stbi__get8(s); + palette[i*4+3] = 255; + } + break; + } + + case STBI__PNG_TYPE('t','R','N','S'): { + if (first) return stbi__err("first not IHDR", "Corrupt PNG"); + if (z->idata) return stbi__err("tRNS after IDAT","Corrupt PNG"); + if (pal_img_n) { + if (scan == STBI__SCAN_header) { s->img_n = 4; return 1; } + if (pal_len == 0) return stbi__err("tRNS before PLTE","Corrupt PNG"); + if (c.length > pal_len) return stbi__err("bad tRNS len","Corrupt PNG"); + pal_img_n = 4; + for (i=0; i < c.length; ++i) + palette[i*4+3] = stbi__get8(s); + } else { + if (!(s->img_n & 1)) return stbi__err("tRNS with alpha","Corrupt PNG"); + if (c.length != (stbi__uint32) s->img_n*2) return stbi__err("bad tRNS len","Corrupt PNG"); + has_trans = 1; + // non-paletted with tRNS = constant alpha. if header-scanning, we can stop now. + if (scan == STBI__SCAN_header) { ++s->img_n; return 1; } + if (z->depth == 16) { + for (k = 0; k < s->img_n && k < 3; ++k) // extra loop test to suppress false GCC warning + tc16[k] = (stbi__uint16)stbi__get16be(s); // copy the values as-is + } else { + for (k = 0; k < s->img_n && k < 3; ++k) + tc[k] = (stbi_uc)(stbi__get16be(s) & 255) * stbi__depth_scale_table[z->depth]; // non 8-bit images will be larger + } + } + break; + } + + case STBI__PNG_TYPE('I','D','A','T'): { + if (first) return stbi__err("first not IHDR", "Corrupt PNG"); + if (pal_img_n && !pal_len) return stbi__err("no PLTE","Corrupt PNG"); + if (scan == STBI__SCAN_header) { + // header scan definitely stops at first IDAT + if (pal_img_n) + s->img_n = pal_img_n; + return 1; + } + if (c.length > (1u << 30)) return stbi__err("IDAT size limit", "IDAT section larger than 2^30 bytes"); + if ((int)(ioff + c.length) < (int)ioff) return 0; + if (ioff + c.length > idata_limit) { + stbi__uint32 idata_limit_old = idata_limit; + stbi_uc *p; + if (idata_limit == 0) idata_limit = c.length > 4096 ? c.length : 4096; + while (ioff + c.length > idata_limit) + idata_limit *= 2; + STBI_NOTUSED(idata_limit_old); + p = (stbi_uc *) STBI_REALLOC_SIZED(z->idata, idata_limit_old, idata_limit); if (p == NULL) return stbi__err("outofmem", "Out of memory"); + z->idata = p; + } + if (!stbi__getn(s, z->idata+ioff,c.length)) return stbi__err("outofdata","Corrupt PNG"); + ioff += c.length; + break; + } + + case STBI__PNG_TYPE('I','E','N','D'): { + stbi__uint32 raw_len, bpl; + if (first) return stbi__err("first not IHDR", "Corrupt PNG"); + if (scan != STBI__SCAN_load) return 1; + if (z->idata == NULL) return stbi__err("no IDAT","Corrupt PNG"); + // initial guess for decoded data size to avoid unnecessary reallocs + bpl = (s->img_x * z->depth + 7) / 8; // bytes per line, per component + raw_len = bpl * s->img_y * s->img_n /* pixels */ + s->img_y /* filter mode per row */; + z->expanded = (stbi_uc *) stbi_zlib_decode_malloc_guesssize_headerflag((char *) z->idata, ioff, raw_len, (int *) &raw_len, !is_iphone); + if (z->expanded == NULL) return 0; // zlib should set error + STBI_FREE(z->idata); z->idata = NULL; + if ((req_comp == s->img_n+1 && req_comp != 3 && !pal_img_n) || has_trans) + s->img_out_n = s->img_n+1; + else + s->img_out_n = s->img_n; + if (!stbi__create_png_image(z, z->expanded, raw_len, s->img_out_n, z->depth, color, interlace)) return 0; + if (has_trans) { + if (z->depth == 16) { + if (!stbi__compute_transparency16(z, tc16, s->img_out_n)) return 0; + } else { + if (!stbi__compute_transparency(z, tc, s->img_out_n)) return 0; + } + } + if (is_iphone && stbi__de_iphone_flag && s->img_out_n > 2) + stbi__de_iphone(z); + if (pal_img_n) { + // pal_img_n == 3 or 4 + s->img_n = pal_img_n; // record the actual colors we had + s->img_out_n = pal_img_n; + if (req_comp >= 3) s->img_out_n = req_comp; + if (!stbi__expand_png_palette(z, palette, pal_len, s->img_out_n)) + return 0; + } else if (has_trans) { + // non-paletted image with tRNS -> source image has (constant) alpha + ++s->img_n; + } + STBI_FREE(z->expanded); z->expanded = NULL; + // end of PNG chunk, read and skip CRC + stbi__get32be(s); + return 1; + } + + default: + // if critical, fail + if (first) return stbi__err("first not IHDR", "Corrupt PNG"); + if ((c.type & (1 << 29)) == 0) { + #ifndef STBI_NO_FAILURE_STRINGS + // not threadsafe + static char invalid_chunk[] = "XXXX PNG chunk not known"; + invalid_chunk[0] = STBI__BYTECAST(c.type >> 24); + invalid_chunk[1] = STBI__BYTECAST(c.type >> 16); + invalid_chunk[2] = STBI__BYTECAST(c.type >> 8); + invalid_chunk[3] = STBI__BYTECAST(c.type >> 0); + #endif + return stbi__err(invalid_chunk, "PNG not supported: unknown PNG chunk type"); + } + stbi__skip(s, c.length); + break; + } + // end of PNG chunk, read and skip CRC + stbi__get32be(s); + } +} + +static void *stbi__do_png(stbi__png *p, int *x, int *y, int *n, int req_comp, stbi__result_info *ri) +{ + void *result=NULL; + if (req_comp < 0 || req_comp > 4) return stbi__errpuc("bad req_comp", "Internal error"); + if (stbi__parse_png_file(p, STBI__SCAN_load, req_comp)) { + if (p->depth <= 8) + ri->bits_per_channel = 8; + else if (p->depth == 16) + ri->bits_per_channel = 16; + else + return stbi__errpuc("bad bits_per_channel", "PNG not supported: unsupported color depth"); + result = p->out; + p->out = NULL; + if (req_comp && req_comp != p->s->img_out_n) { + if (ri->bits_per_channel == 8) + result = stbi__convert_format((unsigned char *) result, p->s->img_out_n, req_comp, p->s->img_x, p->s->img_y); + else + result = stbi__convert_format16((stbi__uint16 *) result, p->s->img_out_n, req_comp, p->s->img_x, p->s->img_y); + p->s->img_out_n = req_comp; + if (result == NULL) return result; + } + *x = p->s->img_x; + *y = p->s->img_y; + if (n) *n = p->s->img_n; + } + STBI_FREE(p->out); p->out = NULL; + STBI_FREE(p->expanded); p->expanded = NULL; + STBI_FREE(p->idata); p->idata = NULL; + + return result; +} + +static void *stbi__png_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + stbi__png p; + p.s = s; + return stbi__do_png(&p, x,y,comp,req_comp, ri); +} + +static int stbi__png_test(stbi__context *s) +{ + int r; + r = stbi__check_png_header(s); + stbi__rewind(s); + return r; +} + +static int stbi__png_info_raw(stbi__png *p, int *x, int *y, int *comp) +{ + if (!stbi__parse_png_file(p, STBI__SCAN_header, 0)) { + stbi__rewind( p->s ); + return 0; + } + if (x) *x = p->s->img_x; + if (y) *y = p->s->img_y; + if (comp) *comp = p->s->img_n; + return 1; +} + +static int stbi__png_info(stbi__context *s, int *x, int *y, int *comp) +{ + stbi__png p; + p.s = s; + return stbi__png_info_raw(&p, x, y, comp); +} + +static int stbi__png_is16(stbi__context *s) +{ + stbi__png p; + p.s = s; + if (!stbi__png_info_raw(&p, NULL, NULL, NULL)) + return 0; + if (p.depth != 16) { + stbi__rewind(p.s); + return 0; + } + return 1; +} +#endif + +// Microsoft/Windows BMP image + +#ifndef STBI_NO_BMP +static int stbi__bmp_test_raw(stbi__context *s) +{ + int r; + int sz; + if (stbi__get8(s) != 'B') return 0; + if (stbi__get8(s) != 'M') return 0; + stbi__get32le(s); // discard filesize + stbi__get16le(s); // discard reserved + stbi__get16le(s); // discard reserved + stbi__get32le(s); // discard data offset + sz = stbi__get32le(s); + r = (sz == 12 || sz == 40 || sz == 56 || sz == 108 || sz == 124); + return r; +} + +static int stbi__bmp_test(stbi__context *s) +{ + int r = stbi__bmp_test_raw(s); + stbi__rewind(s); + return r; +} + + +// returns 0..31 for the highest set bit +static int stbi__high_bit(unsigned int z) +{ + int n=0; + if (z == 0) return -1; + if (z >= 0x10000) { n += 16; z >>= 16; } + if (z >= 0x00100) { n += 8; z >>= 8; } + if (z >= 0x00010) { n += 4; z >>= 4; } + if (z >= 0x00004) { n += 2; z >>= 2; } + if (z >= 0x00002) { n += 1;/* >>= 1;*/ } + return n; +} + +static int stbi__bitcount(unsigned int a) +{ + a = (a & 0x55555555) + ((a >> 1) & 0x55555555); // max 2 + a = (a & 0x33333333) + ((a >> 2) & 0x33333333); // max 4 + a = (a + (a >> 4)) & 0x0f0f0f0f; // max 8 per 4, now 8 bits + a = (a + (a >> 8)); // max 16 per 8 bits + a = (a + (a >> 16)); // max 32 per 8 bits + return a & 0xff; +} + +// extract an arbitrarily-aligned N-bit value (N=bits) +// from v, and then make it 8-bits long and fractionally +// extend it to full full range. +static int stbi__shiftsigned(unsigned int v, int shift, int bits) +{ + static unsigned int mul_table[9] = { + 0, + 0xff/*0b11111111*/, 0x55/*0b01010101*/, 0x49/*0b01001001*/, 0x11/*0b00010001*/, + 0x21/*0b00100001*/, 0x41/*0b01000001*/, 0x81/*0b10000001*/, 0x01/*0b00000001*/, + }; + static unsigned int shift_table[9] = { + 0, 0,0,1,0,2,4,6,0, + }; + if (shift < 0) + v <<= -shift; + else + v >>= shift; + STBI_ASSERT(v < 256); + v >>= (8-bits); + STBI_ASSERT(bits >= 0 && bits <= 8); + return (int) ((unsigned) v * mul_table[bits]) >> shift_table[bits]; +} + +typedef struct +{ + int bpp, offset, hsz; + unsigned int mr,mg,mb,ma, all_a; + int extra_read; +} stbi__bmp_data; + +static int stbi__bmp_set_mask_defaults(stbi__bmp_data *info, int compress) +{ + // BI_BITFIELDS specifies masks explicitly, don't override + if (compress == 3) + return 1; + + if (compress == 0) { + if (info->bpp == 16) { + info->mr = 31u << 10; + info->mg = 31u << 5; + info->mb = 31u << 0; + } else if (info->bpp == 32) { + info->mr = 0xffu << 16; + info->mg = 0xffu << 8; + info->mb = 0xffu << 0; + info->ma = 0xffu << 24; + info->all_a = 0; // if all_a is 0 at end, then we loaded alpha channel but it was all 0 + } else { + // otherwise, use defaults, which is all-0 + info->mr = info->mg = info->mb = info->ma = 0; + } + return 1; + } + return 0; // error +} + +static void *stbi__bmp_parse_header(stbi__context *s, stbi__bmp_data *info) +{ + int hsz; + if (stbi__get8(s) != 'B' || stbi__get8(s) != 'M') return stbi__errpuc("not BMP", "Corrupt BMP"); + stbi__get32le(s); // discard filesize + stbi__get16le(s); // discard reserved + stbi__get16le(s); // discard reserved + info->offset = stbi__get32le(s); + info->hsz = hsz = stbi__get32le(s); + info->mr = info->mg = info->mb = info->ma = 0; + info->extra_read = 14; + + if (info->offset < 0) return stbi__errpuc("bad BMP", "bad BMP"); + + if (hsz != 12 && hsz != 40 && hsz != 56 && hsz != 108 && hsz != 124) return stbi__errpuc("unknown BMP", "BMP type not supported: unknown"); + if (hsz == 12) { + s->img_x = stbi__get16le(s); + s->img_y = stbi__get16le(s); + } else { + s->img_x = stbi__get32le(s); + s->img_y = stbi__get32le(s); + } + if (stbi__get16le(s) != 1) return stbi__errpuc("bad BMP", "bad BMP"); + info->bpp = stbi__get16le(s); + if (hsz != 12) { + int compress = stbi__get32le(s); + if (compress == 1 || compress == 2) return stbi__errpuc("BMP RLE", "BMP type not supported: RLE"); + if (compress >= 4) return stbi__errpuc("BMP JPEG/PNG", "BMP type not supported: unsupported compression"); // this includes PNG/JPEG modes + if (compress == 3 && info->bpp != 16 && info->bpp != 32) return stbi__errpuc("bad BMP", "bad BMP"); // bitfields requires 16 or 32 bits/pixel + stbi__get32le(s); // discard sizeof + stbi__get32le(s); // discard hres + stbi__get32le(s); // discard vres + stbi__get32le(s); // discard colorsused + stbi__get32le(s); // discard max important + if (hsz == 40 || hsz == 56) { + if (hsz == 56) { + stbi__get32le(s); + stbi__get32le(s); + stbi__get32le(s); + stbi__get32le(s); + } + if (info->bpp == 16 || info->bpp == 32) { + if (compress == 0) { + stbi__bmp_set_mask_defaults(info, compress); + } else if (compress == 3) { + info->mr = stbi__get32le(s); + info->mg = stbi__get32le(s); + info->mb = stbi__get32le(s); + info->extra_read += 12; + // not documented, but generated by photoshop and handled by mspaint + if (info->mr == info->mg && info->mg == info->mb) { + // ?!?!? + return stbi__errpuc("bad BMP", "bad BMP"); + } + } else + return stbi__errpuc("bad BMP", "bad BMP"); + } + } else { + // V4/V5 header + int i; + if (hsz != 108 && hsz != 124) + return stbi__errpuc("bad BMP", "bad BMP"); + info->mr = stbi__get32le(s); + info->mg = stbi__get32le(s); + info->mb = stbi__get32le(s); + info->ma = stbi__get32le(s); + if (compress != 3) // override mr/mg/mb unless in BI_BITFIELDS mode, as per docs + stbi__bmp_set_mask_defaults(info, compress); + stbi__get32le(s); // discard color space + for (i=0; i < 12; ++i) + stbi__get32le(s); // discard color space parameters + if (hsz == 124) { + stbi__get32le(s); // discard rendering intent + stbi__get32le(s); // discard offset of profile data + stbi__get32le(s); // discard size of profile data + stbi__get32le(s); // discard reserved + } + } + } + return (void *) 1; +} + + +static void *stbi__bmp_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + stbi_uc *out; + unsigned int mr=0,mg=0,mb=0,ma=0, all_a; + stbi_uc pal[256][4]; + int psize=0,i,j,width; + int flip_vertically, pad, target; + stbi__bmp_data info; + STBI_NOTUSED(ri); + + info.all_a = 255; + if (stbi__bmp_parse_header(s, &info) == NULL) + return NULL; // error code already set + + flip_vertically = ((int) s->img_y) > 0; + s->img_y = abs((int) s->img_y); + + if (s->img_y > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + if (s->img_x > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + + mr = info.mr; + mg = info.mg; + mb = info.mb; + ma = info.ma; + all_a = info.all_a; + + if (info.hsz == 12) { + if (info.bpp < 24) + psize = (info.offset - info.extra_read - 24) / 3; + } else { + if (info.bpp < 16) + psize = (info.offset - info.extra_read - info.hsz) >> 2; + } + if (psize == 0) { + // accept some number of extra bytes after the header, but if the offset points either to before + // the header ends or implies a large amount of extra data, reject the file as malformed + int bytes_read_so_far = s->callback_already_read + (int)(s->img_buffer - s->img_buffer_original); + int header_limit = 1024; // max we actually read is below 256 bytes currently. + int extra_data_limit = 256*4; // what ordinarily goes here is a palette; 256 entries*4 bytes is its max size. + if (bytes_read_so_far <= 0 || bytes_read_so_far > header_limit) { + return stbi__errpuc("bad header", "Corrupt BMP"); + } + // we established that bytes_read_so_far is positive and sensible. + // the first half of this test rejects offsets that are either too small positives, or + // negative, and guarantees that info.offset >= bytes_read_so_far > 0. this in turn + // ensures the number computed in the second half of the test can't overflow. + if (info.offset < bytes_read_so_far || info.offset - bytes_read_so_far > extra_data_limit) { + return stbi__errpuc("bad offset", "Corrupt BMP"); + } else { + stbi__skip(s, info.offset - bytes_read_so_far); + } + } + + if (info.bpp == 24 && ma == 0xff000000) + s->img_n = 3; + else + s->img_n = ma ? 4 : 3; + if (req_comp && req_comp >= 3) // we can directly decode 3 or 4 + target = req_comp; + else + target = s->img_n; // if they want monochrome, we'll post-convert + + // sanity-check size + if (!stbi__mad3sizes_valid(target, s->img_x, s->img_y, 0)) + return stbi__errpuc("too large", "Corrupt BMP"); + + out = (stbi_uc *) stbi__malloc_mad3(target, s->img_x, s->img_y, 0); + if (!out) return stbi__errpuc("outofmem", "Out of memory"); + if (info.bpp < 16) { + int z=0; + if (psize == 0 || psize > 256) { STBI_FREE(out); return stbi__errpuc("invalid", "Corrupt BMP"); } + for (i=0; i < psize; ++i) { + pal[i][2] = stbi__get8(s); + pal[i][1] = stbi__get8(s); + pal[i][0] = stbi__get8(s); + if (info.hsz != 12) stbi__get8(s); + pal[i][3] = 255; + } + stbi__skip(s, info.offset - info.extra_read - info.hsz - psize * (info.hsz == 12 ? 3 : 4)); + if (info.bpp == 1) width = (s->img_x + 7) >> 3; + else if (info.bpp == 4) width = (s->img_x + 1) >> 1; + else if (info.bpp == 8) width = s->img_x; + else { STBI_FREE(out); return stbi__errpuc("bad bpp", "Corrupt BMP"); } + pad = (-width)&3; + if (info.bpp == 1) { + for (j=0; j < (int) s->img_y; ++j) { + int bit_offset = 7, v = stbi__get8(s); + for (i=0; i < (int) s->img_x; ++i) { + int color = (v>>bit_offset)&0x1; + out[z++] = pal[color][0]; + out[z++] = pal[color][1]; + out[z++] = pal[color][2]; + if (target == 4) out[z++] = 255; + if (i+1 == (int) s->img_x) break; + if((--bit_offset) < 0) { + bit_offset = 7; + v = stbi__get8(s); + } + } + stbi__skip(s, pad); + } + } else { + for (j=0; j < (int) s->img_y; ++j) { + for (i=0; i < (int) s->img_x; i += 2) { + int v=stbi__get8(s),v2=0; + if (info.bpp == 4) { + v2 = v & 15; + v >>= 4; + } + out[z++] = pal[v][0]; + out[z++] = pal[v][1]; + out[z++] = pal[v][2]; + if (target == 4) out[z++] = 255; + if (i+1 == (int) s->img_x) break; + v = (info.bpp == 8) ? stbi__get8(s) : v2; + out[z++] = pal[v][0]; + out[z++] = pal[v][1]; + out[z++] = pal[v][2]; + if (target == 4) out[z++] = 255; + } + stbi__skip(s, pad); + } + } + } else { + int rshift=0,gshift=0,bshift=0,ashift=0,rcount=0,gcount=0,bcount=0,acount=0; + int z = 0; + int easy=0; + stbi__skip(s, info.offset - info.extra_read - info.hsz); + if (info.bpp == 24) width = 3 * s->img_x; + else if (info.bpp == 16) width = 2*s->img_x; + else /* bpp = 32 and pad = 0 */ width=0; + pad = (-width) & 3; + if (info.bpp == 24) { + easy = 1; + } else if (info.bpp == 32) { + if (mb == 0xff && mg == 0xff00 && mr == 0x00ff0000 && ma == 0xff000000) + easy = 2; + } + if (!easy) { + if (!mr || !mg || !mb) { STBI_FREE(out); return stbi__errpuc("bad masks", "Corrupt BMP"); } + // right shift amt to put high bit in position #7 + rshift = stbi__high_bit(mr)-7; rcount = stbi__bitcount(mr); + gshift = stbi__high_bit(mg)-7; gcount = stbi__bitcount(mg); + bshift = stbi__high_bit(mb)-7; bcount = stbi__bitcount(mb); + ashift = stbi__high_bit(ma)-7; acount = stbi__bitcount(ma); + if (rcount > 8 || gcount > 8 || bcount > 8 || acount > 8) { STBI_FREE(out); return stbi__errpuc("bad masks", "Corrupt BMP"); } + } + for (j=0; j < (int) s->img_y; ++j) { + if (easy) { + for (i=0; i < (int) s->img_x; ++i) { + unsigned char a; + out[z+2] = stbi__get8(s); + out[z+1] = stbi__get8(s); + out[z+0] = stbi__get8(s); + z += 3; + a = (easy == 2 ? stbi__get8(s) : 255); + all_a |= a; + if (target == 4) out[z++] = a; + } + } else { + int bpp = info.bpp; + for (i=0; i < (int) s->img_x; ++i) { + stbi__uint32 v = (bpp == 16 ? (stbi__uint32) stbi__get16le(s) : stbi__get32le(s)); + unsigned int a; + out[z++] = STBI__BYTECAST(stbi__shiftsigned(v & mr, rshift, rcount)); + out[z++] = STBI__BYTECAST(stbi__shiftsigned(v & mg, gshift, gcount)); + out[z++] = STBI__BYTECAST(stbi__shiftsigned(v & mb, bshift, bcount)); + a = (ma ? stbi__shiftsigned(v & ma, ashift, acount) : 255); + all_a |= a; + if (target == 4) out[z++] = STBI__BYTECAST(a); + } + } + stbi__skip(s, pad); + } + } + + // if alpha channel is all 0s, replace with all 255s + if (target == 4 && all_a == 0) + for (i=4*s->img_x*s->img_y-1; i >= 0; i -= 4) + out[i] = 255; + + if (flip_vertically) { + stbi_uc t; + for (j=0; j < (int) s->img_y>>1; ++j) { + stbi_uc *p1 = out + j *s->img_x*target; + stbi_uc *p2 = out + (s->img_y-1-j)*s->img_x*target; + for (i=0; i < (int) s->img_x*target; ++i) { + t = p1[i]; p1[i] = p2[i]; p2[i] = t; + } + } + } + + if (req_comp && req_comp != target) { + out = stbi__convert_format(out, target, req_comp, s->img_x, s->img_y); + if (out == NULL) return out; // stbi__convert_format frees input on failure + } + + *x = s->img_x; + *y = s->img_y; + if (comp) *comp = s->img_n; + return out; +} +#endif + +// Targa Truevision - TGA +// by Jonathan Dummer +#ifndef STBI_NO_TGA +// returns STBI_rgb or whatever, 0 on error +static int stbi__tga_get_comp(int bits_per_pixel, int is_grey, int* is_rgb16) +{ + // only RGB or RGBA (incl. 16bit) or grey allowed + if (is_rgb16) *is_rgb16 = 0; + switch(bits_per_pixel) { + case 8: return STBI_grey; + case 16: if(is_grey) return STBI_grey_alpha; + // fallthrough + case 15: if(is_rgb16) *is_rgb16 = 1; + return STBI_rgb; + case 24: // fallthrough + case 32: return bits_per_pixel/8; + default: return 0; + } +} + +static int stbi__tga_info(stbi__context *s, int *x, int *y, int *comp) +{ + int tga_w, tga_h, tga_comp, tga_image_type, tga_bits_per_pixel, tga_colormap_bpp; + int sz, tga_colormap_type; + stbi__get8(s); // discard Offset + tga_colormap_type = stbi__get8(s); // colormap type + if( tga_colormap_type > 1 ) { + stbi__rewind(s); + return 0; // only RGB or indexed allowed + } + tga_image_type = stbi__get8(s); // image type + if ( tga_colormap_type == 1 ) { // colormapped (paletted) image + if (tga_image_type != 1 && tga_image_type != 9) { + stbi__rewind(s); + return 0; + } + stbi__skip(s,4); // skip index of first colormap entry and number of entries + sz = stbi__get8(s); // check bits per palette color entry + if ( (sz != 8) && (sz != 15) && (sz != 16) && (sz != 24) && (sz != 32) ) { + stbi__rewind(s); + return 0; + } + stbi__skip(s,4); // skip image x and y origin + tga_colormap_bpp = sz; + } else { // "normal" image w/o colormap - only RGB or grey allowed, +/- RLE + if ( (tga_image_type != 2) && (tga_image_type != 3) && (tga_image_type != 10) && (tga_image_type != 11) ) { + stbi__rewind(s); + return 0; // only RGB or grey allowed, +/- RLE + } + stbi__skip(s,9); // skip colormap specification and image x/y origin + tga_colormap_bpp = 0; + } + tga_w = stbi__get16le(s); + if( tga_w < 1 ) { + stbi__rewind(s); + return 0; // test width + } + tga_h = stbi__get16le(s); + if( tga_h < 1 ) { + stbi__rewind(s); + return 0; // test height + } + tga_bits_per_pixel = stbi__get8(s); // bits per pixel + stbi__get8(s); // ignore alpha bits + if (tga_colormap_bpp != 0) { + if((tga_bits_per_pixel != 8) && (tga_bits_per_pixel != 16)) { + // when using a colormap, tga_bits_per_pixel is the size of the indexes + // I don't think anything but 8 or 16bit indexes makes sense + stbi__rewind(s); + return 0; + } + tga_comp = stbi__tga_get_comp(tga_colormap_bpp, 0, NULL); + } else { + tga_comp = stbi__tga_get_comp(tga_bits_per_pixel, (tga_image_type == 3) || (tga_image_type == 11), NULL); + } + if(!tga_comp) { + stbi__rewind(s); + return 0; + } + if (x) *x = tga_w; + if (y) *y = tga_h; + if (comp) *comp = tga_comp; + return 1; // seems to have passed everything +} + +static int stbi__tga_test(stbi__context *s) +{ + int res = 0; + int sz, tga_color_type; + stbi__get8(s); // discard Offset + tga_color_type = stbi__get8(s); // color type + if ( tga_color_type > 1 ) goto errorEnd; // only RGB or indexed allowed + sz = stbi__get8(s); // image type + if ( tga_color_type == 1 ) { // colormapped (paletted) image + if (sz != 1 && sz != 9) goto errorEnd; // colortype 1 demands image type 1 or 9 + stbi__skip(s,4); // skip index of first colormap entry and number of entries + sz = stbi__get8(s); // check bits per palette color entry + if ( (sz != 8) && (sz != 15) && (sz != 16) && (sz != 24) && (sz != 32) ) goto errorEnd; + stbi__skip(s,4); // skip image x and y origin + } else { // "normal" image w/o colormap + if ( (sz != 2) && (sz != 3) && (sz != 10) && (sz != 11) ) goto errorEnd; // only RGB or grey allowed, +/- RLE + stbi__skip(s,9); // skip colormap specification and image x/y origin + } + if ( stbi__get16le(s) < 1 ) goto errorEnd; // test width + if ( stbi__get16le(s) < 1 ) goto errorEnd; // test height + sz = stbi__get8(s); // bits per pixel + if ( (tga_color_type == 1) && (sz != 8) && (sz != 16) ) goto errorEnd; // for colormapped images, bpp is size of an index + if ( (sz != 8) && (sz != 15) && (sz != 16) && (sz != 24) && (sz != 32) ) goto errorEnd; + + res = 1; // if we got this far, everything's good and we can return 1 instead of 0 + +errorEnd: + stbi__rewind(s); + return res; +} + +// read 16bit value and convert to 24bit RGB +static void stbi__tga_read_rgb16(stbi__context *s, stbi_uc* out) +{ + stbi__uint16 px = (stbi__uint16)stbi__get16le(s); + stbi__uint16 fiveBitMask = 31; + // we have 3 channels with 5bits each + int r = (px >> 10) & fiveBitMask; + int g = (px >> 5) & fiveBitMask; + int b = px & fiveBitMask; + // Note that this saves the data in RGB(A) order, so it doesn't need to be swapped later + out[0] = (stbi_uc)((r * 255)/31); + out[1] = (stbi_uc)((g * 255)/31); + out[2] = (stbi_uc)((b * 255)/31); + + // some people claim that the most significant bit might be used for alpha + // (possibly if an alpha-bit is set in the "image descriptor byte") + // but that only made 16bit test images completely translucent.. + // so let's treat all 15 and 16bit TGAs as RGB with no alpha. +} + +static void *stbi__tga_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + // read in the TGA header stuff + int tga_offset = stbi__get8(s); + int tga_indexed = stbi__get8(s); + int tga_image_type = stbi__get8(s); + int tga_is_RLE = 0; + int tga_palette_start = stbi__get16le(s); + int tga_palette_len = stbi__get16le(s); + int tga_palette_bits = stbi__get8(s); + int tga_x_origin = stbi__get16le(s); + int tga_y_origin = stbi__get16le(s); + int tga_width = stbi__get16le(s); + int tga_height = stbi__get16le(s); + int tga_bits_per_pixel = stbi__get8(s); + int tga_comp, tga_rgb16=0; + int tga_inverted = stbi__get8(s); + // int tga_alpha_bits = tga_inverted & 15; // the 4 lowest bits - unused (useless?) + // image data + unsigned char *tga_data; + unsigned char *tga_palette = NULL; + int i, j; + unsigned char raw_data[4] = {0}; + int RLE_count = 0; + int RLE_repeating = 0; + int read_next_pixel = 1; + STBI_NOTUSED(ri); + STBI_NOTUSED(tga_x_origin); // @TODO + STBI_NOTUSED(tga_y_origin); // @TODO + + if (tga_height > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + if (tga_width > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + + // do a tiny bit of precessing + if ( tga_image_type >= 8 ) + { + tga_image_type -= 8; + tga_is_RLE = 1; + } + tga_inverted = 1 - ((tga_inverted >> 5) & 1); + + // If I'm paletted, then I'll use the number of bits from the palette + if ( tga_indexed ) tga_comp = stbi__tga_get_comp(tga_palette_bits, 0, &tga_rgb16); + else tga_comp = stbi__tga_get_comp(tga_bits_per_pixel, (tga_image_type == 3), &tga_rgb16); + + if(!tga_comp) // shouldn't really happen, stbi__tga_test() should have ensured basic consistency + return stbi__errpuc("bad format", "Can't find out TGA pixelformat"); + + // tga info + *x = tga_width; + *y = tga_height; + if (comp) *comp = tga_comp; + + if (!stbi__mad3sizes_valid(tga_width, tga_height, tga_comp, 0)) + return stbi__errpuc("too large", "Corrupt TGA"); + + tga_data = (unsigned char*)stbi__malloc_mad3(tga_width, tga_height, tga_comp, 0); + if (!tga_data) return stbi__errpuc("outofmem", "Out of memory"); + + // skip to the data's starting position (offset usually = 0) + stbi__skip(s, tga_offset ); + + if ( !tga_indexed && !tga_is_RLE && !tga_rgb16 ) { + for (i=0; i < tga_height; ++i) { + int row = tga_inverted ? tga_height -i - 1 : i; + stbi_uc *tga_row = tga_data + row*tga_width*tga_comp; + stbi__getn(s, tga_row, tga_width * tga_comp); + } + } else { + // do I need to load a palette? + if ( tga_indexed) + { + if (tga_palette_len == 0) { /* you have to have at least one entry! */ + STBI_FREE(tga_data); + return stbi__errpuc("bad palette", "Corrupt TGA"); + } + + // any data to skip? (offset usually = 0) + stbi__skip(s, tga_palette_start ); + // load the palette + tga_palette = (unsigned char*)stbi__malloc_mad2(tga_palette_len, tga_comp, 0); + if (!tga_palette) { + STBI_FREE(tga_data); + return stbi__errpuc("outofmem", "Out of memory"); + } + if (tga_rgb16) { + stbi_uc *pal_entry = tga_palette; + STBI_ASSERT(tga_comp == STBI_rgb); + for (i=0; i < tga_palette_len; ++i) { + stbi__tga_read_rgb16(s, pal_entry); + pal_entry += tga_comp; + } + } else if (!stbi__getn(s, tga_palette, tga_palette_len * tga_comp)) { + STBI_FREE(tga_data); + STBI_FREE(tga_palette); + return stbi__errpuc("bad palette", "Corrupt TGA"); + } + } + // load the data + for (i=0; i < tga_width * tga_height; ++i) + { + // if I'm in RLE mode, do I need to get a RLE stbi__pngchunk? + if ( tga_is_RLE ) + { + if ( RLE_count == 0 ) + { + // yep, get the next byte as a RLE command + int RLE_cmd = stbi__get8(s); + RLE_count = 1 + (RLE_cmd & 127); + RLE_repeating = RLE_cmd >> 7; + read_next_pixel = 1; + } else if ( !RLE_repeating ) + { + read_next_pixel = 1; + } + } else + { + read_next_pixel = 1; + } + // OK, if I need to read a pixel, do it now + if ( read_next_pixel ) + { + // load however much data we did have + if ( tga_indexed ) + { + // read in index, then perform the lookup + int pal_idx = (tga_bits_per_pixel == 8) ? stbi__get8(s) : stbi__get16le(s); + if ( pal_idx >= tga_palette_len ) { + // invalid index + pal_idx = 0; + } + pal_idx *= tga_comp; + for (j = 0; j < tga_comp; ++j) { + raw_data[j] = tga_palette[pal_idx+j]; + } + } else if(tga_rgb16) { + STBI_ASSERT(tga_comp == STBI_rgb); + stbi__tga_read_rgb16(s, raw_data); + } else { + // read in the data raw + for (j = 0; j < tga_comp; ++j) { + raw_data[j] = stbi__get8(s); + } + } + // clear the reading flag for the next pixel + read_next_pixel = 0; + } // end of reading a pixel + + // copy data + for (j = 0; j < tga_comp; ++j) + tga_data[i*tga_comp+j] = raw_data[j]; + + // in case we're in RLE mode, keep counting down + --RLE_count; + } + // do I need to invert the image? + if ( tga_inverted ) + { + for (j = 0; j*2 < tga_height; ++j) + { + int index1 = j * tga_width * tga_comp; + int index2 = (tga_height - 1 - j) * tga_width * tga_comp; + for (i = tga_width * tga_comp; i > 0; --i) + { + unsigned char temp = tga_data[index1]; + tga_data[index1] = tga_data[index2]; + tga_data[index2] = temp; + ++index1; + ++index2; + } + } + } + // clear my palette, if I had one + if ( tga_palette != NULL ) + { + STBI_FREE( tga_palette ); + } + } + + // swap RGB - if the source data was RGB16, it already is in the right order + if (tga_comp >= 3 && !tga_rgb16) + { + unsigned char* tga_pixel = tga_data; + for (i=0; i < tga_width * tga_height; ++i) + { + unsigned char temp = tga_pixel[0]; + tga_pixel[0] = tga_pixel[2]; + tga_pixel[2] = temp; + tga_pixel += tga_comp; + } + } + + // convert to target component count + if (req_comp && req_comp != tga_comp) + tga_data = stbi__convert_format(tga_data, tga_comp, req_comp, tga_width, tga_height); + + // the things I do to get rid of an error message, and yet keep + // Microsoft's C compilers happy... [8^( + tga_palette_start = tga_palette_len = tga_palette_bits = + tga_x_origin = tga_y_origin = 0; + STBI_NOTUSED(tga_palette_start); + // OK, done + return tga_data; +} +#endif + +// ************************************************************************************************* +// Photoshop PSD loader -- PD by Thatcher Ulrich, integration by Nicolas Schulz, tweaked by STB + +#ifndef STBI_NO_PSD +static int stbi__psd_test(stbi__context *s) +{ + int r = (stbi__get32be(s) == 0x38425053); + stbi__rewind(s); + return r; +} + +static int stbi__psd_decode_rle(stbi__context *s, stbi_uc *p, int pixelCount) +{ + int count, nleft, len; + + count = 0; + while ((nleft = pixelCount - count) > 0) { + len = stbi__get8(s); + if (len == 128) { + // No-op. + } else if (len < 128) { + // Copy next len+1 bytes literally. + len++; + if (len > nleft) return 0; // corrupt data + count += len; + while (len) { + *p = stbi__get8(s); + p += 4; + len--; + } + } else if (len > 128) { + stbi_uc val; + // Next -len+1 bytes in the dest are replicated from next source byte. + // (Interpret len as a negative 8-bit int.) + len = 257 - len; + if (len > nleft) return 0; // corrupt data + val = stbi__get8(s); + count += len; + while (len) { + *p = val; + p += 4; + len--; + } + } + } + + return 1; +} + +static void *stbi__psd_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri, int bpc) +{ + int pixelCount; + int channelCount, compression; + int channel, i; + int bitdepth; + int w,h; + stbi_uc *out; + STBI_NOTUSED(ri); + + // Check identifier + if (stbi__get32be(s) != 0x38425053) // "8BPS" + return stbi__errpuc("not PSD", "Corrupt PSD image"); + + // Check file type version. + if (stbi__get16be(s) != 1) + return stbi__errpuc("wrong version", "Unsupported version of PSD image"); + + // Skip 6 reserved bytes. + stbi__skip(s, 6 ); + + // Read the number of channels (R, G, B, A, etc). + channelCount = stbi__get16be(s); + if (channelCount < 0 || channelCount > 16) + return stbi__errpuc("wrong channel count", "Unsupported number of channels in PSD image"); + + // Read the rows and columns of the image. + h = stbi__get32be(s); + w = stbi__get32be(s); + + if (h > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + if (w > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + + // Make sure the depth is 8 bits. + bitdepth = stbi__get16be(s); + if (bitdepth != 8 && bitdepth != 16) + return stbi__errpuc("unsupported bit depth", "PSD bit depth is not 8 or 16 bit"); + + // Make sure the color mode is RGB. + // Valid options are: + // 0: Bitmap + // 1: Grayscale + // 2: Indexed color + // 3: RGB color + // 4: CMYK color + // 7: Multichannel + // 8: Duotone + // 9: Lab color + if (stbi__get16be(s) != 3) + return stbi__errpuc("wrong color format", "PSD is not in RGB color format"); + + // Skip the Mode Data. (It's the palette for indexed color; other info for other modes.) + stbi__skip(s,stbi__get32be(s) ); + + // Skip the image resources. (resolution, pen tool paths, etc) + stbi__skip(s, stbi__get32be(s) ); + + // Skip the reserved data. + stbi__skip(s, stbi__get32be(s) ); + + // Find out if the data is compressed. + // Known values: + // 0: no compression + // 1: RLE compressed + compression = stbi__get16be(s); + if (compression > 1) + return stbi__errpuc("bad compression", "PSD has an unknown compression format"); + + // Check size + if (!stbi__mad3sizes_valid(4, w, h, 0)) + return stbi__errpuc("too large", "Corrupt PSD"); + + // Create the destination image. + + if (!compression && bitdepth == 16 && bpc == 16) { + out = (stbi_uc *) stbi__malloc_mad3(8, w, h, 0); + ri->bits_per_channel = 16; + } else + out = (stbi_uc *) stbi__malloc(4 * w*h); + + if (!out) return stbi__errpuc("outofmem", "Out of memory"); + pixelCount = w*h; + + // Initialize the data to zero. + //memset( out, 0, pixelCount * 4 ); + + // Finally, the image data. + if (compression) { + // RLE as used by .PSD and .TIFF + // Loop until you get the number of unpacked bytes you are expecting: + // Read the next source byte into n. + // If n is between 0 and 127 inclusive, copy the next n+1 bytes literally. + // Else if n is between -127 and -1 inclusive, copy the next byte -n+1 times. + // Else if n is 128, noop. + // Endloop + + // The RLE-compressed data is preceded by a 2-byte data count for each row in the data, + // which we're going to just skip. + stbi__skip(s, h * channelCount * 2 ); + + // Read the RLE data by channel. + for (channel = 0; channel < 4; channel++) { + stbi_uc *p; + + p = out+channel; + if (channel >= channelCount) { + // Fill this channel with default data. + for (i = 0; i < pixelCount; i++, p += 4) + *p = (channel == 3 ? 255 : 0); + } else { + // Read the RLE data. + if (!stbi__psd_decode_rle(s, p, pixelCount)) { + STBI_FREE(out); + return stbi__errpuc("corrupt", "bad RLE data"); + } + } + } + + } else { + // We're at the raw image data. It's each channel in order (Red, Green, Blue, Alpha, ...) + // where each channel consists of an 8-bit (or 16-bit) value for each pixel in the image. + + // Read the data by channel. + for (channel = 0; channel < 4; channel++) { + if (channel >= channelCount) { + // Fill this channel with default data. + if (bitdepth == 16 && bpc == 16) { + stbi__uint16 *q = ((stbi__uint16 *) out) + channel; + stbi__uint16 val = channel == 3 ? 65535 : 0; + for (i = 0; i < pixelCount; i++, q += 4) + *q = val; + } else { + stbi_uc *p = out+channel; + stbi_uc val = channel == 3 ? 255 : 0; + for (i = 0; i < pixelCount; i++, p += 4) + *p = val; + } + } else { + if (ri->bits_per_channel == 16) { // output bpc + stbi__uint16 *q = ((stbi__uint16 *) out) + channel; + for (i = 0; i < pixelCount; i++, q += 4) + *q = (stbi__uint16) stbi__get16be(s); + } else { + stbi_uc *p = out+channel; + if (bitdepth == 16) { // input bpc + for (i = 0; i < pixelCount; i++, p += 4) + *p = (stbi_uc) (stbi__get16be(s) >> 8); + } else { + for (i = 0; i < pixelCount; i++, p += 4) + *p = stbi__get8(s); + } + } + } + } + } + + // remove weird white matte from PSD + if (channelCount >= 4) { + if (ri->bits_per_channel == 16) { + for (i=0; i < w*h; ++i) { + stbi__uint16 *pixel = (stbi__uint16 *) out + 4*i; + if (pixel[3] != 0 && pixel[3] != 65535) { + float a = pixel[3] / 65535.0f; + float ra = 1.0f / a; + float inv_a = 65535.0f * (1 - ra); + pixel[0] = (stbi__uint16) (pixel[0]*ra + inv_a); + pixel[1] = (stbi__uint16) (pixel[1]*ra + inv_a); + pixel[2] = (stbi__uint16) (pixel[2]*ra + inv_a); + } + } + } else { + for (i=0; i < w*h; ++i) { + unsigned char *pixel = out + 4*i; + if (pixel[3] != 0 && pixel[3] != 255) { + float a = pixel[3] / 255.0f; + float ra = 1.0f / a; + float inv_a = 255.0f * (1 - ra); + pixel[0] = (unsigned char) (pixel[0]*ra + inv_a); + pixel[1] = (unsigned char) (pixel[1]*ra + inv_a); + pixel[2] = (unsigned char) (pixel[2]*ra + inv_a); + } + } + } + } + + // convert to desired output format + if (req_comp && req_comp != 4) { + if (ri->bits_per_channel == 16) + out = (stbi_uc *) stbi__convert_format16((stbi__uint16 *) out, 4, req_comp, w, h); + else + out = stbi__convert_format(out, 4, req_comp, w, h); + if (out == NULL) return out; // stbi__convert_format frees input on failure + } + + if (comp) *comp = 4; + *y = h; + *x = w; + + return out; +} +#endif + +// ************************************************************************************************* +// Softimage PIC loader +// by Tom Seddon +// +// See http://softimage.wiki.softimage.com/index.php/INFO:_PIC_file_format +// See http://ozviz.wasp.uwa.edu.au/~pbourke/dataformats/softimagepic/ + +#ifndef STBI_NO_PIC +static int stbi__pic_is4(stbi__context *s,const char *str) +{ + int i; + for (i=0; i<4; ++i) + if (stbi__get8(s) != (stbi_uc)str[i]) + return 0; + + return 1; +} + +static int stbi__pic_test_core(stbi__context *s) +{ + int i; + + if (!stbi__pic_is4(s,"\x53\x80\xF6\x34")) + return 0; + + for(i=0;i<84;++i) + stbi__get8(s); + + if (!stbi__pic_is4(s,"PICT")) + return 0; + + return 1; +} + +typedef struct +{ + stbi_uc size,type,channel; +} stbi__pic_packet; + +static stbi_uc *stbi__readval(stbi__context *s, int channel, stbi_uc *dest) +{ + int mask=0x80, i; + + for (i=0; i<4; ++i, mask>>=1) { + if (channel & mask) { + if (stbi__at_eof(s)) return stbi__errpuc("bad file","PIC file too short"); + dest[i]=stbi__get8(s); + } + } + + return dest; +} + +static void stbi__copyval(int channel,stbi_uc *dest,const stbi_uc *src) +{ + int mask=0x80,i; + + for (i=0;i<4; ++i, mask>>=1) + if (channel&mask) + dest[i]=src[i]; +} + +static stbi_uc *stbi__pic_load_core(stbi__context *s,int width,int height,int *comp, stbi_uc *result) +{ + int act_comp=0,num_packets=0,y,chained; + stbi__pic_packet packets[10]; + + // this will (should...) cater for even some bizarre stuff like having data + // for the same channel in multiple packets. + do { + stbi__pic_packet *packet; + + if (num_packets==sizeof(packets)/sizeof(packets[0])) + return stbi__errpuc("bad format","too many packets"); + + packet = &packets[num_packets++]; + + chained = stbi__get8(s); + packet->size = stbi__get8(s); + packet->type = stbi__get8(s); + packet->channel = stbi__get8(s); + + act_comp |= packet->channel; + + if (stbi__at_eof(s)) return stbi__errpuc("bad file","file too short (reading packets)"); + if (packet->size != 8) return stbi__errpuc("bad format","packet isn't 8bpp"); + } while (chained); + + *comp = (act_comp & 0x10 ? 4 : 3); // has alpha channel? + + for(y=0; ytype) { + default: + return stbi__errpuc("bad format","packet has bad compression type"); + + case 0: {//uncompressed + int x; + + for(x=0;xchannel,dest)) + return 0; + break; + } + + case 1://Pure RLE + { + int left=width, i; + + while (left>0) { + stbi_uc count,value[4]; + + count=stbi__get8(s); + if (stbi__at_eof(s)) return stbi__errpuc("bad file","file too short (pure read count)"); + + if (count > left) + count = (stbi_uc) left; + + if (!stbi__readval(s,packet->channel,value)) return 0; + + for(i=0; ichannel,dest,value); + left -= count; + } + } + break; + + case 2: {//Mixed RLE + int left=width; + while (left>0) { + int count = stbi__get8(s), i; + if (stbi__at_eof(s)) return stbi__errpuc("bad file","file too short (mixed read count)"); + + if (count >= 128) { // Repeated + stbi_uc value[4]; + + if (count==128) + count = stbi__get16be(s); + else + count -= 127; + if (count > left) + return stbi__errpuc("bad file","scanline overrun"); + + if (!stbi__readval(s,packet->channel,value)) + return 0; + + for(i=0;ichannel,dest,value); + } else { // Raw + ++count; + if (count>left) return stbi__errpuc("bad file","scanline overrun"); + + for(i=0;ichannel,dest)) + return 0; + } + left-=count; + } + break; + } + } + } + } + + return result; +} + +static void *stbi__pic_load(stbi__context *s,int *px,int *py,int *comp,int req_comp, stbi__result_info *ri) +{ + stbi_uc *result; + int i, x,y, internal_comp; + STBI_NOTUSED(ri); + + if (!comp) comp = &internal_comp; + + for (i=0; i<92; ++i) + stbi__get8(s); + + x = stbi__get16be(s); + y = stbi__get16be(s); + + if (y > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + if (x > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + + if (stbi__at_eof(s)) return stbi__errpuc("bad file","file too short (pic header)"); + if (!stbi__mad3sizes_valid(x, y, 4, 0)) return stbi__errpuc("too large", "PIC image too large to decode"); + + stbi__get32be(s); //skip `ratio' + stbi__get16be(s); //skip `fields' + stbi__get16be(s); //skip `pad' + + // intermediate buffer is RGBA + result = (stbi_uc *) stbi__malloc_mad3(x, y, 4, 0); + if (!result) return stbi__errpuc("outofmem", "Out of memory"); + memset(result, 0xff, x*y*4); + + if (!stbi__pic_load_core(s,x,y,comp, result)) { + STBI_FREE(result); + result=0; + } + *px = x; + *py = y; + if (req_comp == 0) req_comp = *comp; + result=stbi__convert_format(result,4,req_comp,x,y); + + return result; +} + +static int stbi__pic_test(stbi__context *s) +{ + int r = stbi__pic_test_core(s); + stbi__rewind(s); + return r; +} +#endif + +// ************************************************************************************************* +// GIF loader -- public domain by Jean-Marc Lienher -- simplified/shrunk by stb + +#ifndef STBI_NO_GIF +typedef struct +{ + stbi__int16 prefix; + stbi_uc first; + stbi_uc suffix; +} stbi__gif_lzw; + +typedef struct +{ + int w,h; + stbi_uc *out; // output buffer (always 4 components) + stbi_uc *background; // The current "background" as far as a gif is concerned + stbi_uc *history; + int flags, bgindex, ratio, transparent, eflags; + stbi_uc pal[256][4]; + stbi_uc lpal[256][4]; + stbi__gif_lzw codes[8192]; + stbi_uc *color_table; + int parse, step; + int lflags; + int start_x, start_y; + int max_x, max_y; + int cur_x, cur_y; + int line_size; + int delay; +} stbi__gif; + +static int stbi__gif_test_raw(stbi__context *s) +{ + int sz; + if (stbi__get8(s) != 'G' || stbi__get8(s) != 'I' || stbi__get8(s) != 'F' || stbi__get8(s) != '8') return 0; + sz = stbi__get8(s); + if (sz != '9' && sz != '7') return 0; + if (stbi__get8(s) != 'a') return 0; + return 1; +} + +static int stbi__gif_test(stbi__context *s) +{ + int r = stbi__gif_test_raw(s); + stbi__rewind(s); + return r; +} + +static void stbi__gif_parse_colortable(stbi__context *s, stbi_uc pal[256][4], int num_entries, int transp) +{ + int i; + for (i=0; i < num_entries; ++i) { + pal[i][2] = stbi__get8(s); + pal[i][1] = stbi__get8(s); + pal[i][0] = stbi__get8(s); + pal[i][3] = transp == i ? 0 : 255; + } +} + +static int stbi__gif_header(stbi__context *s, stbi__gif *g, int *comp, int is_info) +{ + stbi_uc version; + if (stbi__get8(s) != 'G' || stbi__get8(s) != 'I' || stbi__get8(s) != 'F' || stbi__get8(s) != '8') + return stbi__err("not GIF", "Corrupt GIF"); + + version = stbi__get8(s); + if (version != '7' && version != '9') return stbi__err("not GIF", "Corrupt GIF"); + if (stbi__get8(s) != 'a') return stbi__err("not GIF", "Corrupt GIF"); + + stbi__g_failure_reason = ""; + g->w = stbi__get16le(s); + g->h = stbi__get16le(s); + g->flags = stbi__get8(s); + g->bgindex = stbi__get8(s); + g->ratio = stbi__get8(s); + g->transparent = -1; + + if (g->w > STBI_MAX_DIMENSIONS) return stbi__err("too large","Very large image (corrupt?)"); + if (g->h > STBI_MAX_DIMENSIONS) return stbi__err("too large","Very large image (corrupt?)"); + + if (comp != 0) *comp = 4; // can't actually tell whether it's 3 or 4 until we parse the comments + + if (is_info) return 1; + + if (g->flags & 0x80) + stbi__gif_parse_colortable(s,g->pal, 2 << (g->flags & 7), -1); + + return 1; +} + +static int stbi__gif_info_raw(stbi__context *s, int *x, int *y, int *comp) +{ + stbi__gif* g = (stbi__gif*) stbi__malloc(sizeof(stbi__gif)); + if (!g) return stbi__err("outofmem", "Out of memory"); + if (!stbi__gif_header(s, g, comp, 1)) { + STBI_FREE(g); + stbi__rewind( s ); + return 0; + } + if (x) *x = g->w; + if (y) *y = g->h; + STBI_FREE(g); + return 1; +} + +static void stbi__out_gif_code(stbi__gif *g, stbi__uint16 code) +{ + stbi_uc *p, *c; + int idx; + + // recurse to decode the prefixes, since the linked-list is backwards, + // and working backwards through an interleaved image would be nasty + if (g->codes[code].prefix >= 0) + stbi__out_gif_code(g, g->codes[code].prefix); + + if (g->cur_y >= g->max_y) return; + + idx = g->cur_x + g->cur_y; + p = &g->out[idx]; + g->history[idx / 4] = 1; + + c = &g->color_table[g->codes[code].suffix * 4]; + if (c[3] > 128) { // don't render transparent pixels; + p[0] = c[2]; + p[1] = c[1]; + p[2] = c[0]; + p[3] = c[3]; + } + g->cur_x += 4; + + if (g->cur_x >= g->max_x) { + g->cur_x = g->start_x; + g->cur_y += g->step; + + while (g->cur_y >= g->max_y && g->parse > 0) { + g->step = (1 << g->parse) * g->line_size; + g->cur_y = g->start_y + (g->step >> 1); + --g->parse; + } + } +} + +static stbi_uc *stbi__process_gif_raster(stbi__context *s, stbi__gif *g) +{ + stbi_uc lzw_cs; + stbi__int32 len, init_code; + stbi__uint32 first; + stbi__int32 codesize, codemask, avail, oldcode, bits, valid_bits, clear; + stbi__gif_lzw *p; + + lzw_cs = stbi__get8(s); + if (lzw_cs > 12) return NULL; + clear = 1 << lzw_cs; + first = 1; + codesize = lzw_cs + 1; + codemask = (1 << codesize) - 1; + bits = 0; + valid_bits = 0; + for (init_code = 0; init_code < clear; init_code++) { + g->codes[init_code].prefix = -1; + g->codes[init_code].first = (stbi_uc) init_code; + g->codes[init_code].suffix = (stbi_uc) init_code; + } + + // support no starting clear code + avail = clear+2; + oldcode = -1; + + len = 0; + for(;;) { + if (valid_bits < codesize) { + if (len == 0) { + len = stbi__get8(s); // start new block + if (len == 0) + return g->out; + } + --len; + bits |= (stbi__int32) stbi__get8(s) << valid_bits; + valid_bits += 8; + } else { + stbi__int32 code = bits & codemask; + bits >>= codesize; + valid_bits -= codesize; + // @OPTIMIZE: is there some way we can accelerate the non-clear path? + if (code == clear) { // clear code + codesize = lzw_cs + 1; + codemask = (1 << codesize) - 1; + avail = clear + 2; + oldcode = -1; + first = 0; + } else if (code == clear + 1) { // end of stream code + stbi__skip(s, len); + while ((len = stbi__get8(s)) > 0) + stbi__skip(s,len); + return g->out; + } else if (code <= avail) { + if (first) { + return stbi__errpuc("no clear code", "Corrupt GIF"); + } + + if (oldcode >= 0) { + p = &g->codes[avail++]; + if (avail > 8192) { + return stbi__errpuc("too many codes", "Corrupt GIF"); + } + + p->prefix = (stbi__int16) oldcode; + p->first = g->codes[oldcode].first; + p->suffix = (code == avail) ? p->first : g->codes[code].first; + } else if (code == avail) + return stbi__errpuc("illegal code in raster", "Corrupt GIF"); + + stbi__out_gif_code(g, (stbi__uint16) code); + + if ((avail & codemask) == 0 && avail <= 0x0FFF) { + codesize++; + codemask = (1 << codesize) - 1; + } + + oldcode = code; + } else { + return stbi__errpuc("illegal code in raster", "Corrupt GIF"); + } + } + } +} + +// this function is designed to support animated gifs, although stb_image doesn't support it +// two back is the image from two frames ago, used for a very specific disposal format +static stbi_uc *stbi__gif_load_next(stbi__context *s, stbi__gif *g, int *comp, int req_comp, stbi_uc *two_back) +{ + int dispose; + int first_frame; + int pi; + int pcount; + STBI_NOTUSED(req_comp); + + // on first frame, any non-written pixels get the background colour (non-transparent) + first_frame = 0; + if (g->out == 0) { + if (!stbi__gif_header(s, g, comp,0)) return 0; // stbi__g_failure_reason set by stbi__gif_header + if (!stbi__mad3sizes_valid(4, g->w, g->h, 0)) + return stbi__errpuc("too large", "GIF image is too large"); + pcount = g->w * g->h; + g->out = (stbi_uc *) stbi__malloc(4 * pcount); + g->background = (stbi_uc *) stbi__malloc(4 * pcount); + g->history = (stbi_uc *) stbi__malloc(pcount); + if (!g->out || !g->background || !g->history) + return stbi__errpuc("outofmem", "Out of memory"); + + // image is treated as "transparent" at the start - ie, nothing overwrites the current background; + // background colour is only used for pixels that are not rendered first frame, after that "background" + // color refers to the color that was there the previous frame. + memset(g->out, 0x00, 4 * pcount); + memset(g->background, 0x00, 4 * pcount); // state of the background (starts transparent) + memset(g->history, 0x00, pcount); // pixels that were affected previous frame + first_frame = 1; + } else { + // second frame - how do we dispose of the previous one? + dispose = (g->eflags & 0x1C) >> 2; + pcount = g->w * g->h; + + if ((dispose == 3) && (two_back == 0)) { + dispose = 2; // if I don't have an image to revert back to, default to the old background + } + + if (dispose == 3) { // use previous graphic + for (pi = 0; pi < pcount; ++pi) { + if (g->history[pi]) { + memcpy( &g->out[pi * 4], &two_back[pi * 4], 4 ); + } + } + } else if (dispose == 2) { + // restore what was changed last frame to background before that frame; + for (pi = 0; pi < pcount; ++pi) { + if (g->history[pi]) { + memcpy( &g->out[pi * 4], &g->background[pi * 4], 4 ); + } + } + } else { + // This is a non-disposal case eithe way, so just + // leave the pixels as is, and they will become the new background + // 1: do not dispose + // 0: not specified. + } + + // background is what out is after the undoing of the previou frame; + memcpy( g->background, g->out, 4 * g->w * g->h ); + } + + // clear my history; + memset( g->history, 0x00, g->w * g->h ); // pixels that were affected previous frame + + for (;;) { + int tag = stbi__get8(s); + switch (tag) { + case 0x2C: /* Image Descriptor */ + { + stbi__int32 x, y, w, h; + stbi_uc *o; + + x = stbi__get16le(s); + y = stbi__get16le(s); + w = stbi__get16le(s); + h = stbi__get16le(s); + if (((x + w) > (g->w)) || ((y + h) > (g->h))) + return stbi__errpuc("bad Image Descriptor", "Corrupt GIF"); + + g->line_size = g->w * 4; + g->start_x = x * 4; + g->start_y = y * g->line_size; + g->max_x = g->start_x + w * 4; + g->max_y = g->start_y + h * g->line_size; + g->cur_x = g->start_x; + g->cur_y = g->start_y; + + // if the width of the specified rectangle is 0, that means + // we may not see *any* pixels or the image is malformed; + // to make sure this is caught, move the current y down to + // max_y (which is what out_gif_code checks). + if (w == 0) + g->cur_y = g->max_y; + + g->lflags = stbi__get8(s); + + if (g->lflags & 0x40) { + g->step = 8 * g->line_size; // first interlaced spacing + g->parse = 3; + } else { + g->step = g->line_size; + g->parse = 0; + } + + if (g->lflags & 0x80) { + stbi__gif_parse_colortable(s,g->lpal, 2 << (g->lflags & 7), g->eflags & 0x01 ? g->transparent : -1); + g->color_table = (stbi_uc *) g->lpal; + } else if (g->flags & 0x80) { + g->color_table = (stbi_uc *) g->pal; + } else + return stbi__errpuc("missing color table", "Corrupt GIF"); + + o = stbi__process_gif_raster(s, g); + if (!o) return NULL; + + // if this was the first frame, + pcount = g->w * g->h; + if (first_frame && (g->bgindex > 0)) { + // if first frame, any pixel not drawn to gets the background color + for (pi = 0; pi < pcount; ++pi) { + if (g->history[pi] == 0) { + g->pal[g->bgindex][3] = 255; // just in case it was made transparent, undo that; It will be reset next frame if need be; + memcpy( &g->out[pi * 4], &g->pal[g->bgindex], 4 ); + } + } + } + + return o; + } + + case 0x21: // Comment Extension. + { + int len; + int ext = stbi__get8(s); + if (ext == 0xF9) { // Graphic Control Extension. + len = stbi__get8(s); + if (len == 4) { + g->eflags = stbi__get8(s); + g->delay = 10 * stbi__get16le(s); // delay - 1/100th of a second, saving as 1/1000ths. + + // unset old transparent + if (g->transparent >= 0) { + g->pal[g->transparent][3] = 255; + } + if (g->eflags & 0x01) { + g->transparent = stbi__get8(s); + if (g->transparent >= 0) { + g->pal[g->transparent][3] = 0; + } + } else { + // don't need transparent + stbi__skip(s, 1); + g->transparent = -1; + } + } else { + stbi__skip(s, len); + break; + } + } + while ((len = stbi__get8(s)) != 0) { + stbi__skip(s, len); + } + break; + } + + case 0x3B: // gif stream termination code + return (stbi_uc *) s; // using '1' causes warning on some compilers + + default: + return stbi__errpuc("unknown code", "Corrupt GIF"); + } + } +} + +static void *stbi__load_gif_main_outofmem(stbi__gif *g, stbi_uc *out, int **delays) +{ + STBI_FREE(g->out); + STBI_FREE(g->history); + STBI_FREE(g->background); + + if (out) STBI_FREE(out); + if (delays && *delays) STBI_FREE(*delays); + return stbi__errpuc("outofmem", "Out of memory"); +} + +static void *stbi__load_gif_main(stbi__context *s, int **delays, int *x, int *y, int *z, int *comp, int req_comp) +{ + if (stbi__gif_test(s)) { + int layers = 0; + stbi_uc *u = 0; + stbi_uc *out = 0; + stbi_uc *two_back = 0; + stbi__gif g; + int stride; + int out_size = 0; + int delays_size = 0; + + STBI_NOTUSED(out_size); + STBI_NOTUSED(delays_size); + + memset(&g, 0, sizeof(g)); + if (delays) { + *delays = 0; + } + + do { + u = stbi__gif_load_next(s, &g, comp, req_comp, two_back); + if (u == (stbi_uc *) s) u = 0; // end of animated gif marker + + if (u) { + *x = g.w; + *y = g.h; + ++layers; + stride = g.w * g.h * 4; + + if (out) { + void *tmp = (stbi_uc*) STBI_REALLOC_SIZED( out, out_size, layers * stride ); + if (!tmp) + return stbi__load_gif_main_outofmem(&g, out, delays); + else { + out = (stbi_uc*) tmp; + out_size = layers * stride; + } + + if (delays) { + int *new_delays = (int*) STBI_REALLOC_SIZED( *delays, delays_size, sizeof(int) * layers ); + if (!new_delays) + return stbi__load_gif_main_outofmem(&g, out, delays); + *delays = new_delays; + delays_size = layers * sizeof(int); + } + } else { + out = (stbi_uc*)stbi__malloc( layers * stride ); + if (!out) + return stbi__load_gif_main_outofmem(&g, out, delays); + out_size = layers * stride; + if (delays) { + *delays = (int*) stbi__malloc( layers * sizeof(int) ); + if (!*delays) + return stbi__load_gif_main_outofmem(&g, out, delays); + delays_size = layers * sizeof(int); + } + } + memcpy( out + ((layers - 1) * stride), u, stride ); + if (layers >= 2) { + two_back = out - 2 * stride; + } + + if (delays) { + (*delays)[layers - 1U] = g.delay; + } + } + } while (u != 0); + + // free temp buffer; + STBI_FREE(g.out); + STBI_FREE(g.history); + STBI_FREE(g.background); + + // do the final conversion after loading everything; + if (req_comp && req_comp != 4) + out = stbi__convert_format(out, 4, req_comp, layers * g.w, g.h); + + *z = layers; + return out; + } else { + return stbi__errpuc("not GIF", "Image was not as a gif type."); + } +} + +static void *stbi__gif_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + stbi_uc *u = 0; + stbi__gif g; + memset(&g, 0, sizeof(g)); + STBI_NOTUSED(ri); + + u = stbi__gif_load_next(s, &g, comp, req_comp, 0); + if (u == (stbi_uc *) s) u = 0; // end of animated gif marker + if (u) { + *x = g.w; + *y = g.h; + + // moved conversion to after successful load so that the same + // can be done for multiple frames. + if (req_comp && req_comp != 4) + u = stbi__convert_format(u, 4, req_comp, g.w, g.h); + } else if (g.out) { + // if there was an error and we allocated an image buffer, free it! + STBI_FREE(g.out); + } + + // free buffers needed for multiple frame loading; + STBI_FREE(g.history); + STBI_FREE(g.background); + + return u; +} + +static int stbi__gif_info(stbi__context *s, int *x, int *y, int *comp) +{ + return stbi__gif_info_raw(s,x,y,comp); +} +#endif + +// ************************************************************************************************* +// Radiance RGBE HDR loader +// originally by Nicolas Schulz +#ifndef STBI_NO_HDR +static int stbi__hdr_test_core(stbi__context *s, const char *signature) +{ + int i; + for (i=0; signature[i]; ++i) + if (stbi__get8(s) != signature[i]) + return 0; + stbi__rewind(s); + return 1; +} + +static int stbi__hdr_test(stbi__context* s) +{ + int r = stbi__hdr_test_core(s, "#?RADIANCE\n"); + stbi__rewind(s); + if(!r) { + r = stbi__hdr_test_core(s, "#?RGBE\n"); + stbi__rewind(s); + } + return r; +} + +#define STBI__HDR_BUFLEN 1024 +static char *stbi__hdr_gettoken(stbi__context *z, char *buffer) +{ + int len=0; + char c = '\0'; + + c = (char) stbi__get8(z); + + while (!stbi__at_eof(z) && c != '\n') { + buffer[len++] = c; + if (len == STBI__HDR_BUFLEN-1) { + // flush to end of line + while (!stbi__at_eof(z) && stbi__get8(z) != '\n') + ; + break; + } + c = (char) stbi__get8(z); + } + + buffer[len] = 0; + return buffer; +} + +static void stbi__hdr_convert(float *output, stbi_uc *input, int req_comp) +{ + if ( input[3] != 0 ) { + float f1; + // Exponent + f1 = (float) ldexp(1.0f, input[3] - (int)(128 + 8)); + if (req_comp <= 2) + output[0] = (input[0] + input[1] + input[2]) * f1 / 3; + else { + output[0] = input[0] * f1; + output[1] = input[1] * f1; + output[2] = input[2] * f1; + } + if (req_comp == 2) output[1] = 1; + if (req_comp == 4) output[3] = 1; + } else { + switch (req_comp) { + case 4: output[3] = 1; /* fallthrough */ + case 3: output[0] = output[1] = output[2] = 0; + break; + case 2: output[1] = 1; /* fallthrough */ + case 1: output[0] = 0; + break; + } + } +} + +static float *stbi__hdr_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + char buffer[STBI__HDR_BUFLEN]; + char *token; + int valid = 0; + int width, height; + stbi_uc *scanline; + float *hdr_data; + int len; + unsigned char count, value; + int i, j, k, c1,c2, z; + const char *headerToken; + STBI_NOTUSED(ri); + + // Check identifier + headerToken = stbi__hdr_gettoken(s,buffer); + if (strcmp(headerToken, "#?RADIANCE") != 0 && strcmp(headerToken, "#?RGBE") != 0) + return stbi__errpf("not HDR", "Corrupt HDR image"); + + // Parse header + for(;;) { + token = stbi__hdr_gettoken(s,buffer); + if (token[0] == 0) break; + if (strcmp(token, "FORMAT=32-bit_rle_rgbe") == 0) valid = 1; + } + + if (!valid) return stbi__errpf("unsupported format", "Unsupported HDR format"); + + // Parse width and height + // can't use sscanf() if we're not using stdio! + token = stbi__hdr_gettoken(s,buffer); + if (strncmp(token, "-Y ", 3)) return stbi__errpf("unsupported data layout", "Unsupported HDR format"); + token += 3; + height = (int) strtol(token, &token, 10); + while (*token == ' ') ++token; + if (strncmp(token, "+X ", 3)) return stbi__errpf("unsupported data layout", "Unsupported HDR format"); + token += 3; + width = (int) strtol(token, NULL, 10); + + if (height > STBI_MAX_DIMENSIONS) return stbi__errpf("too large","Very large image (corrupt?)"); + if (width > STBI_MAX_DIMENSIONS) return stbi__errpf("too large","Very large image (corrupt?)"); + + *x = width; + *y = height; + + if (comp) *comp = 3; + if (req_comp == 0) req_comp = 3; + + if (!stbi__mad4sizes_valid(width, height, req_comp, sizeof(float), 0)) + return stbi__errpf("too large", "HDR image is too large"); + + // Read data + hdr_data = (float *) stbi__malloc_mad4(width, height, req_comp, sizeof(float), 0); + if (!hdr_data) + return stbi__errpf("outofmem", "Out of memory"); + + // Load image data + // image data is stored as some number of sca + if ( width < 8 || width >= 32768) { + // Read flat data + for (j=0; j < height; ++j) { + for (i=0; i < width; ++i) { + stbi_uc rgbe[4]; + main_decode_loop: + stbi__getn(s, rgbe, 4); + stbi__hdr_convert(hdr_data + j * width * req_comp + i * req_comp, rgbe, req_comp); + } + } + } else { + // Read RLE-encoded data + scanline = NULL; + + for (j = 0; j < height; ++j) { + c1 = stbi__get8(s); + c2 = stbi__get8(s); + len = stbi__get8(s); + if (c1 != 2 || c2 != 2 || (len & 0x80)) { + // not run-length encoded, so we have to actually use THIS data as a decoded + // pixel (note this can't be a valid pixel--one of RGB must be >= 128) + stbi_uc rgbe[4]; + rgbe[0] = (stbi_uc) c1; + rgbe[1] = (stbi_uc) c2; + rgbe[2] = (stbi_uc) len; + rgbe[3] = (stbi_uc) stbi__get8(s); + stbi__hdr_convert(hdr_data, rgbe, req_comp); + i = 1; + j = 0; + STBI_FREE(scanline); + goto main_decode_loop; // yes, this makes no sense + } + len <<= 8; + len |= stbi__get8(s); + if (len != width) { STBI_FREE(hdr_data); STBI_FREE(scanline); return stbi__errpf("invalid decoded scanline length", "corrupt HDR"); } + if (scanline == NULL) { + scanline = (stbi_uc *) stbi__malloc_mad2(width, 4, 0); + if (!scanline) { + STBI_FREE(hdr_data); + return stbi__errpf("outofmem", "Out of memory"); + } + } + + for (k = 0; k < 4; ++k) { + int nleft; + i = 0; + while ((nleft = width - i) > 0) { + count = stbi__get8(s); + if (count > 128) { + // Run + value = stbi__get8(s); + count -= 128; + if ((count == 0) || (count > nleft)) { STBI_FREE(hdr_data); STBI_FREE(scanline); return stbi__errpf("corrupt", "bad RLE data in HDR"); } + for (z = 0; z < count; ++z) + scanline[i++ * 4 + k] = value; + } else { + // Dump + if ((count == 0) || (count > nleft)) { STBI_FREE(hdr_data); STBI_FREE(scanline); return stbi__errpf("corrupt", "bad RLE data in HDR"); } + for (z = 0; z < count; ++z) + scanline[i++ * 4 + k] = stbi__get8(s); + } + } + } + for (i=0; i < width; ++i) + stbi__hdr_convert(hdr_data+(j*width + i)*req_comp, scanline + i*4, req_comp); + } + if (scanline) + STBI_FREE(scanline); + } + + return hdr_data; +} + +static int stbi__hdr_info(stbi__context *s, int *x, int *y, int *comp) +{ + char buffer[STBI__HDR_BUFLEN]; + char *token; + int valid = 0; + int dummy; + + if (!x) x = &dummy; + if (!y) y = &dummy; + if (!comp) comp = &dummy; + + if (stbi__hdr_test(s) == 0) { + stbi__rewind( s ); + return 0; + } + + for(;;) { + token = stbi__hdr_gettoken(s,buffer); + if (token[0] == 0) break; + if (strcmp(token, "FORMAT=32-bit_rle_rgbe") == 0) valid = 1; + } + + if (!valid) { + stbi__rewind( s ); + return 0; + } + token = stbi__hdr_gettoken(s,buffer); + if (strncmp(token, "-Y ", 3)) { + stbi__rewind( s ); + return 0; + } + token += 3; + *y = (int) strtol(token, &token, 10); + while (*token == ' ') ++token; + if (strncmp(token, "+X ", 3)) { + stbi__rewind( s ); + return 0; + } + token += 3; + *x = (int) strtol(token, NULL, 10); + *comp = 3; + return 1; +} +#endif // STBI_NO_HDR + +#ifndef STBI_NO_BMP +static int stbi__bmp_info(stbi__context *s, int *x, int *y, int *comp) +{ + void *p; + stbi__bmp_data info; + + info.all_a = 255; + p = stbi__bmp_parse_header(s, &info); + if (p == NULL) { + stbi__rewind( s ); + return 0; + } + if (x) *x = s->img_x; + if (y) *y = s->img_y; + if (comp) { + if (info.bpp == 24 && info.ma == 0xff000000) + *comp = 3; + else + *comp = info.ma ? 4 : 3; + } + return 1; +} +#endif + +#ifndef STBI_NO_PSD +static int stbi__psd_info(stbi__context *s, int *x, int *y, int *comp) +{ + int channelCount, dummy, depth; + if (!x) x = &dummy; + if (!y) y = &dummy; + if (!comp) comp = &dummy; + if (stbi__get32be(s) != 0x38425053) { + stbi__rewind( s ); + return 0; + } + if (stbi__get16be(s) != 1) { + stbi__rewind( s ); + return 0; + } + stbi__skip(s, 6); + channelCount = stbi__get16be(s); + if (channelCount < 0 || channelCount > 16) { + stbi__rewind( s ); + return 0; + } + *y = stbi__get32be(s); + *x = stbi__get32be(s); + depth = stbi__get16be(s); + if (depth != 8 && depth != 16) { + stbi__rewind( s ); + return 0; + } + if (stbi__get16be(s) != 3) { + stbi__rewind( s ); + return 0; + } + *comp = 4; + return 1; +} + +static int stbi__psd_is16(stbi__context *s) +{ + int channelCount, depth; + if (stbi__get32be(s) != 0x38425053) { + stbi__rewind( s ); + return 0; + } + if (stbi__get16be(s) != 1) { + stbi__rewind( s ); + return 0; + } + stbi__skip(s, 6); + channelCount = stbi__get16be(s); + if (channelCount < 0 || channelCount > 16) { + stbi__rewind( s ); + return 0; + } + STBI_NOTUSED(stbi__get32be(s)); + STBI_NOTUSED(stbi__get32be(s)); + depth = stbi__get16be(s); + if (depth != 16) { + stbi__rewind( s ); + return 0; + } + return 1; +} +#endif + +#ifndef STBI_NO_PIC +static int stbi__pic_info(stbi__context *s, int *x, int *y, int *comp) +{ + int act_comp=0,num_packets=0,chained,dummy; + stbi__pic_packet packets[10]; + + if (!x) x = &dummy; + if (!y) y = &dummy; + if (!comp) comp = &dummy; + + if (!stbi__pic_is4(s,"\x53\x80\xF6\x34")) { + stbi__rewind(s); + return 0; + } + + stbi__skip(s, 88); + + *x = stbi__get16be(s); + *y = stbi__get16be(s); + if (stbi__at_eof(s)) { + stbi__rewind( s); + return 0; + } + if ( (*x) != 0 && (1 << 28) / (*x) < (*y)) { + stbi__rewind( s ); + return 0; + } + + stbi__skip(s, 8); + + do { + stbi__pic_packet *packet; + + if (num_packets==sizeof(packets)/sizeof(packets[0])) + return 0; + + packet = &packets[num_packets++]; + chained = stbi__get8(s); + packet->size = stbi__get8(s); + packet->type = stbi__get8(s); + packet->channel = stbi__get8(s); + act_comp |= packet->channel; + + if (stbi__at_eof(s)) { + stbi__rewind( s ); + return 0; + } + if (packet->size != 8) { + stbi__rewind( s ); + return 0; + } + } while (chained); + + *comp = (act_comp & 0x10 ? 4 : 3); + + return 1; +} +#endif + +// ************************************************************************************************* +// Portable Gray Map and Portable Pixel Map loader +// by Ken Miller +// +// PGM: http://netpbm.sourceforge.net/doc/pgm.html +// PPM: http://netpbm.sourceforge.net/doc/ppm.html +// +// Known limitations: +// Does not support comments in the header section +// Does not support ASCII image data (formats P2 and P3) + +#ifndef STBI_NO_PNM + +static int stbi__pnm_test(stbi__context *s) +{ + char p, t; + p = (char) stbi__get8(s); + t = (char) stbi__get8(s); + if (p != 'P' || (t != '5' && t != '6')) { + stbi__rewind( s ); + return 0; + } + return 1; +} + +static void *stbi__pnm_load(stbi__context *s, int *x, int *y, int *comp, int req_comp, stbi__result_info *ri) +{ + stbi_uc *out; + STBI_NOTUSED(ri); + + ri->bits_per_channel = stbi__pnm_info(s, (int *)&s->img_x, (int *)&s->img_y, (int *)&s->img_n); + if (ri->bits_per_channel == 0) + return 0; + + if (s->img_y > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + if (s->img_x > STBI_MAX_DIMENSIONS) return stbi__errpuc("too large","Very large image (corrupt?)"); + + *x = s->img_x; + *y = s->img_y; + if (comp) *comp = s->img_n; + + if (!stbi__mad4sizes_valid(s->img_n, s->img_x, s->img_y, ri->bits_per_channel / 8, 0)) + return stbi__errpuc("too large", "PNM too large"); + + out = (stbi_uc *) stbi__malloc_mad4(s->img_n, s->img_x, s->img_y, ri->bits_per_channel / 8, 0); + if (!out) return stbi__errpuc("outofmem", "Out of memory"); + if (!stbi__getn(s, out, s->img_n * s->img_x * s->img_y * (ri->bits_per_channel / 8))) { + STBI_FREE(out); + return stbi__errpuc("bad PNM", "PNM file truncated"); + } + + if (req_comp && req_comp != s->img_n) { + if (ri->bits_per_channel == 16) { + out = (stbi_uc *) stbi__convert_format16((stbi__uint16 *) out, s->img_n, req_comp, s->img_x, s->img_y); + } else { + out = stbi__convert_format(out, s->img_n, req_comp, s->img_x, s->img_y); + } + if (out == NULL) return out; // stbi__convert_format frees input on failure + } + return out; +} + +static int stbi__pnm_isspace(char c) +{ + return c == ' ' || c == '\t' || c == '\n' || c == '\v' || c == '\f' || c == '\r'; +} + +static void stbi__pnm_skip_whitespace(stbi__context *s, char *c) +{ + for (;;) { + while (!stbi__at_eof(s) && stbi__pnm_isspace(*c)) + *c = (char) stbi__get8(s); + + if (stbi__at_eof(s) || *c != '#') + break; + + while (!stbi__at_eof(s) && *c != '\n' && *c != '\r' ) + *c = (char) stbi__get8(s); + } +} + +static int stbi__pnm_isdigit(char c) +{ + return c >= '0' && c <= '9'; +} + +static int stbi__pnm_getinteger(stbi__context *s, char *c) +{ + int value = 0; + + while (!stbi__at_eof(s) && stbi__pnm_isdigit(*c)) { + value = value*10 + (*c - '0'); + *c = (char) stbi__get8(s); + if((value > 214748364) || (value == 214748364 && *c > '7')) + return stbi__err("integer parse overflow", "Parsing an integer in the PPM header overflowed a 32-bit int"); + } + + return value; +} + +static int stbi__pnm_info(stbi__context *s, int *x, int *y, int *comp) +{ + int maxv, dummy; + char c, p, t; + + if (!x) x = &dummy; + if (!y) y = &dummy; + if (!comp) comp = &dummy; + + stbi__rewind(s); + + // Get identifier + p = (char) stbi__get8(s); + t = (char) stbi__get8(s); + if (p != 'P' || (t != '5' && t != '6')) { + stbi__rewind(s); + return 0; + } + + *comp = (t == '6') ? 3 : 1; // '5' is 1-component .pgm; '6' is 3-component .ppm + + c = (char) stbi__get8(s); + stbi__pnm_skip_whitespace(s, &c); + + *x = stbi__pnm_getinteger(s, &c); // read width + if(*x == 0) + return stbi__err("invalid width", "PPM image header had zero or overflowing width"); + stbi__pnm_skip_whitespace(s, &c); + + *y = stbi__pnm_getinteger(s, &c); // read height + if (*y == 0) + return stbi__err("invalid width", "PPM image header had zero or overflowing width"); + stbi__pnm_skip_whitespace(s, &c); + + maxv = stbi__pnm_getinteger(s, &c); // read max value + if (maxv > 65535) + return stbi__err("max value > 65535", "PPM image supports only 8-bit and 16-bit images"); + else if (maxv > 255) + return 16; + else + return 8; +} + +static int stbi__pnm_is16(stbi__context *s) +{ + if (stbi__pnm_info(s, NULL, NULL, NULL) == 16) + return 1; + return 0; +} +#endif + +static int stbi__info_main(stbi__context *s, int *x, int *y, int *comp) +{ + #ifndef STBI_NO_JPEG + if (stbi__jpeg_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_PNG + if (stbi__png_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_GIF + if (stbi__gif_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_BMP + if (stbi__bmp_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_PSD + if (stbi__psd_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_PIC + if (stbi__pic_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_PNM + if (stbi__pnm_info(s, x, y, comp)) return 1; + #endif + + #ifndef STBI_NO_HDR + if (stbi__hdr_info(s, x, y, comp)) return 1; + #endif + + // test tga last because it's a crappy test! + #ifndef STBI_NO_TGA + if (stbi__tga_info(s, x, y, comp)) + return 1; + #endif + return stbi__err("unknown image type", "Image not of any known type, or corrupt"); +} + +static int stbi__is_16_main(stbi__context *s) +{ + #ifndef STBI_NO_PNG + if (stbi__png_is16(s)) return 1; + #endif + + #ifndef STBI_NO_PSD + if (stbi__psd_is16(s)) return 1; + #endif + + #ifndef STBI_NO_PNM + if (stbi__pnm_is16(s)) return 1; + #endif + return 0; +} + +#ifndef STBI_NO_STDIO +STBIDEF int stbi_info(char const *filename, int *x, int *y, int *comp) +{ + FILE *f = stbi__fopen(filename, "rb"); + int result; + if (!f) return stbi__err("can't fopen", "Unable to open file"); + result = stbi_info_from_file(f, x, y, comp); + fclose(f); + return result; +} + +STBIDEF int stbi_info_from_file(FILE *f, int *x, int *y, int *comp) +{ + int r; + stbi__context s; + long pos = ftell(f); + stbi__start_file(&s, f); + r = stbi__info_main(&s,x,y,comp); + fseek(f,pos,SEEK_SET); + return r; +} + +STBIDEF int stbi_is_16_bit(char const *filename) +{ + FILE *f = stbi__fopen(filename, "rb"); + int result; + if (!f) return stbi__err("can't fopen", "Unable to open file"); + result = stbi_is_16_bit_from_file(f); + fclose(f); + return result; +} + +STBIDEF int stbi_is_16_bit_from_file(FILE *f) +{ + int r; + stbi__context s; + long pos = ftell(f); + stbi__start_file(&s, f); + r = stbi__is_16_main(&s); + fseek(f,pos,SEEK_SET); + return r; +} +#endif // !STBI_NO_STDIO + +STBIDEF int stbi_info_from_memory(stbi_uc const *buffer, int len, int *x, int *y, int *comp) +{ + stbi__context s; + stbi__start_mem(&s,buffer,len); + return stbi__info_main(&s,x,y,comp); +} + +STBIDEF int stbi_info_from_callbacks(stbi_io_callbacks const *c, void *user, int *x, int *y, int *comp) +{ + stbi__context s; + stbi__start_callbacks(&s, (stbi_io_callbacks *) c, user); + return stbi__info_main(&s,x,y,comp); +} + +STBIDEF int stbi_is_16_bit_from_memory(stbi_uc const *buffer, int len) +{ + stbi__context s; + stbi__start_mem(&s,buffer,len); + return stbi__is_16_main(&s); +} + +STBIDEF int stbi_is_16_bit_from_callbacks(stbi_io_callbacks const *c, void *user) +{ + stbi__context s; + stbi__start_callbacks(&s, (stbi_io_callbacks *) c, user); + return stbi__is_16_main(&s); +} + +#endif // STB_IMAGE_IMPLEMENTATION + +/* + revision history: + 2.20 (2019-02-07) support utf8 filenames in Windows; fix warnings and platform ifdefs + 2.19 (2018-02-11) fix warning + 2.18 (2018-01-30) fix warnings + 2.17 (2018-01-29) change sbti__shiftsigned to avoid clang -O2 bug + 1-bit BMP + *_is_16_bit api + avoid warnings + 2.16 (2017-07-23) all functions have 16-bit variants; + STBI_NO_STDIO works again; + compilation fixes; + fix rounding in unpremultiply; + optimize vertical flip; + disable raw_len validation; + documentation fixes + 2.15 (2017-03-18) fix png-1,2,4 bug; now all Imagenet JPGs decode; + warning fixes; disable run-time SSE detection on gcc; + uniform handling of optional "return" values; + thread-safe initialization of zlib tables + 2.14 (2017-03-03) remove deprecated STBI_JPEG_OLD; fixes for Imagenet JPGs + 2.13 (2016-11-29) add 16-bit API, only supported for PNG right now + 2.12 (2016-04-02) fix typo in 2.11 PSD fix that caused crashes + 2.11 (2016-04-02) allocate large structures on the stack + remove white matting for transparent PSD + fix reported channel count for PNG & BMP + re-enable SSE2 in non-gcc 64-bit + support RGB-formatted JPEG + read 16-bit PNGs (only as 8-bit) + 2.10 (2016-01-22) avoid warning introduced in 2.09 by STBI_REALLOC_SIZED + 2.09 (2016-01-16) allow comments in PNM files + 16-bit-per-pixel TGA (not bit-per-component) + info() for TGA could break due to .hdr handling + info() for BMP to shares code instead of sloppy parse + can use STBI_REALLOC_SIZED if allocator doesn't support realloc + code cleanup + 2.08 (2015-09-13) fix to 2.07 cleanup, reading RGB PSD as RGBA + 2.07 (2015-09-13) fix compiler warnings + partial animated GIF support + limited 16-bpc PSD support + #ifdef unused functions + bug with < 92 byte PIC,PNM,HDR,TGA + 2.06 (2015-04-19) fix bug where PSD returns wrong '*comp' value + 2.05 (2015-04-19) fix bug in progressive JPEG handling, fix warning + 2.04 (2015-04-15) try to re-enable SIMD on MinGW 64-bit + 2.03 (2015-04-12) extra corruption checking (mmozeiko) + stbi_set_flip_vertically_on_load (nguillemot) + fix NEON support; fix mingw support + 2.02 (2015-01-19) fix incorrect assert, fix warning + 2.01 (2015-01-17) fix various warnings; suppress SIMD on gcc 32-bit without -msse2 + 2.00b (2014-12-25) fix STBI_MALLOC in progressive JPEG + 2.00 (2014-12-25) optimize JPG, including x86 SSE2 & NEON SIMD (ryg) + progressive JPEG (stb) + PGM/PPM support (Ken Miller) + STBI_MALLOC,STBI_REALLOC,STBI_FREE + GIF bugfix -- seemingly never worked + STBI_NO_*, STBI_ONLY_* + 1.48 (2014-12-14) fix incorrectly-named assert() + 1.47 (2014-12-14) 1/2/4-bit PNG support, both direct and paletted (Omar Cornut & stb) + optimize PNG (ryg) + fix bug in interlaced PNG with user-specified channel count (stb) + 1.46 (2014-08-26) + fix broken tRNS chunk (colorkey-style transparency) in non-paletted PNG + 1.45 (2014-08-16) + fix MSVC-ARM internal compiler error by wrapping malloc + 1.44 (2014-08-07) + various warning fixes from Ronny Chevalier + 1.43 (2014-07-15) + fix MSVC-only compiler problem in code changed in 1.42 + 1.42 (2014-07-09) + don't define _CRT_SECURE_NO_WARNINGS (affects user code) + fixes to stbi__cleanup_jpeg path + added STBI_ASSERT to avoid requiring assert.h + 1.41 (2014-06-25) + fix search&replace from 1.36 that messed up comments/error messages + 1.40 (2014-06-22) + fix gcc struct-initialization warning + 1.39 (2014-06-15) + fix to TGA optimization when req_comp != number of components in TGA; + fix to GIF loading because BMP wasn't rewinding (whoops, no GIFs in my test suite) + add support for BMP version 5 (more ignored fields) + 1.38 (2014-06-06) + suppress MSVC warnings on integer casts truncating values + fix accidental rename of 'skip' field of I/O + 1.37 (2014-06-04) + remove duplicate typedef + 1.36 (2014-06-03) + convert to header file single-file library + if de-iphone isn't set, load iphone images color-swapped instead of returning NULL + 1.35 (2014-05-27) + various warnings + fix broken STBI_SIMD path + fix bug where stbi_load_from_file no longer left file pointer in correct place + fix broken non-easy path for 32-bit BMP (possibly never used) + TGA optimization by Arseny Kapoulkine + 1.34 (unknown) + use STBI_NOTUSED in stbi__resample_row_generic(), fix one more leak in tga failure case + 1.33 (2011-07-14) + make stbi_is_hdr work in STBI_NO_HDR (as specified), minor compiler-friendly improvements + 1.32 (2011-07-13) + support for "info" function for all supported filetypes (SpartanJ) + 1.31 (2011-06-20) + a few more leak fixes, bug in PNG handling (SpartanJ) + 1.30 (2011-06-11) + added ability to load files via callbacks to accomidate custom input streams (Ben Wenger) + removed deprecated format-specific test/load functions + removed support for installable file formats (stbi_loader) -- would have been broken for IO callbacks anyway + error cases in bmp and tga give messages and don't leak (Raymond Barbiero, grisha) + fix inefficiency in decoding 32-bit BMP (David Woo) + 1.29 (2010-08-16) + various warning fixes from Aurelien Pocheville + 1.28 (2010-08-01) + fix bug in GIF palette transparency (SpartanJ) + 1.27 (2010-08-01) + cast-to-stbi_uc to fix warnings + 1.26 (2010-07-24) + fix bug in file buffering for PNG reported by SpartanJ + 1.25 (2010-07-17) + refix trans_data warning (Won Chun) + 1.24 (2010-07-12) + perf improvements reading from files on platforms with lock-heavy fgetc() + minor perf improvements for jpeg + deprecated type-specific functions so we'll get feedback if they're needed + attempt to fix trans_data warning (Won Chun) + 1.23 fixed bug in iPhone support + 1.22 (2010-07-10) + removed image *writing* support + stbi_info support from Jetro Lauha + GIF support from Jean-Marc Lienher + iPhone PNG-extensions from James Brown + warning-fixes from Nicolas Schulz and Janez Zemva (i.stbi__err. Janez (U+017D)emva) + 1.21 fix use of 'stbi_uc' in header (reported by jon blow) + 1.20 added support for Softimage PIC, by Tom Seddon + 1.19 bug in interlaced PNG corruption check (found by ryg) + 1.18 (2008-08-02) + fix a threading bug (local mutable static) + 1.17 support interlaced PNG + 1.16 major bugfix - stbi__convert_format converted one too many pixels + 1.15 initialize some fields for thread safety + 1.14 fix threadsafe conversion bug + header-file-only version (#define STBI_HEADER_FILE_ONLY before including) + 1.13 threadsafe + 1.12 const qualifiers in the API + 1.11 Support installable IDCT, colorspace conversion routines + 1.10 Fixes for 64-bit (don't use "unsigned long") + optimized upsampling by Fabian "ryg" Giesen + 1.09 Fix format-conversion for PSD code (bad global variables!) + 1.08 Thatcher Ulrich's PSD code integrated by Nicolas Schulz + 1.07 attempt to fix C++ warning/errors again + 1.06 attempt to fix C++ warning/errors again + 1.05 fix TGA loading to return correct *comp and use good luminance calc + 1.04 default float alpha is 1, not 255; use 'void *' for stbi_image_free + 1.03 bugfixes to STBI_NO_STDIO, STBI_NO_HDR + 1.02 support for (subset of) HDR files, float interface for preferred access to them + 1.01 fix bug: possible bug in handling right-side up bmps... not sure + fix bug: the stbi__bmp_load() and stbi__tga_load() functions didn't work at all + 1.00 interface to zlib that skips zlib header + 0.99 correct handling of alpha in palette + 0.98 TGA loader by lonesock; dynamically add loaders (untested) + 0.97 jpeg errors on too large a file; also catch another malloc failure + 0.96 fix detection of invalid v value - particleman@mollyrocket forum + 0.95 during header scan, seek to markers in case of padding + 0.94 STBI_NO_STDIO to disable stdio usage; rename all #defines the same + 0.93 handle jpegtran output; verbose errors + 0.92 read 4,8,16,24,32-bit BMP files of several formats + 0.91 output 24-bit Windows 3.0 BMP files + 0.90 fix a few more warnings; bump version number to approach 1.0 + 0.61 bugfixes due to Marc LeBlanc, Christopher Lloyd + 0.60 fix compiling as c++ + 0.59 fix warnings: merge Dave Moore's -Wall fixes + 0.58 fix bug: zlib uncompressed mode len/nlen was wrong endian + 0.57 fix bug: jpg last huffman symbol before marker was >9 bits but less than 16 available + 0.56 fix bug: zlib uncompressed mode len vs. nlen + 0.55 fix bug: restart_interval not initialized to 0 + 0.54 allow NULL for 'int *comp' + 0.53 fix bug in png 3->4; speedup png decoding + 0.52 png handles req_comp=3,4 directly; minor cleanup; jpeg comments + 0.51 obey req_comp requests, 1-component jpegs return as 1-component, + on 'test' only check type, not whether we support this variant + 0.50 (2006-11-19) + first released version +*/ + + +/* +------------------------------------------------------------------------------ +This software is available under 2 licenses -- choose whichever you prefer. +------------------------------------------------------------------------------ +ALTERNATIVE A - MIT License +Copyright (c) 2017 Sean Barrett +Permission is hereby granted, free of charge, to any person obtaining a copy of +this software and associated documentation files (the "Software"), to deal in +the Software without restriction, including without limitation the rights to +use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +of the Software, and to permit persons to whom the Software is furnished to do +so, subject to the following conditions: +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +------------------------------------------------------------------------------ +ALTERNATIVE B - Public Domain (www.unlicense.org) +This is free and unencumbered software released into the public domain. +Anyone is free to copy, modify, publish, use, compile, sell, or distribute this +software, either in source code form or as a compiled binary, for any purpose, +commercial or non-commercial, and by any means. +In jurisdictions that recognize copyright laws, the author or authors of this +software dedicate any and all copyright interest in the software to the public +domain. We make this dedication for the benefit of the public at large and to +the detriment of our heirs and successors. We intend this dedication to be an +overt act of relinquishment in perpetuity of all present and future rights to +this software under copyright law. +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN +ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION +WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +------------------------------------------------------------------------------ +*/ diff --git a/native_render/touch_controller.h b/native_render/touch_controller.h new file mode 100644 index 00000000..62150e84 --- /dev/null +++ b/native_render/touch_controller.h @@ -0,0 +1,732 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include + +#include "UIRenderCommands.h" + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT +#include "platform/ScriptLib/PythonBoot.h" +#endif + +class TouchController { +public: + struct ButtonDef { + int id; // 1 = ATK, 2 = S1, 3 = S2, 4 = S3, 5 = POT, 6 = PICK + // 99 = MENU (toggle drawer) + // 10 = BAG (DIK_I), 11 = CHAR (DIK_C), 12 = SKILL (DIK_V), 13 = QUEST (DIK_N), 14 = COMM (DIK_M), 15 = SET (DIK_ESCAPE) + int dik; + float x, y, radius; + const char* label; + uint32_t color_idle; + uint32_t color_pressed; + bool pressed = false; + int64_t finger_id = -1; + }; + + TouchController() { + init_buttons(); + } + + void set_enabled(bool enabled) { enabled_ = enabled; } + bool is_enabled() const { return enabled_; } + + void update_screen_size(int width, int height) { + if (width <= 0 || height <= 0) return; + screen_w_ = width; + screen_h_ = height; + // Position joystick on lower-left + joystick_base_x_ = std::max(90.0f, float(width) * 0.12f); + joystick_base_y_ = float(height) - std::max(90.0f, float(height) * 0.22f); + if (!joystick_active_) { + joystick_knob_x_ = joystick_base_x_; + joystick_knob_y_ = joystick_base_y_; + } + layout_buttons(); + } + + // Called once per frame to maintain continuous analog movement + void update() { + if (!enabled_) return; + if (joystick_active_) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::SetMoveDirection(move_angle_, true); +#endif + } + } + + // Handles finger touch down (norm_x, norm_y in 0.0 .. 1.0) + bool on_finger_down(int64_t finger_id, float norm_x, float norm_y) { + if (!enabled_) return false; + const float px = norm_x * float(screen_w_); + const float py = norm_y * float(screen_h_); + + // 1. Check HUD buttons (Action cluster & Drawer menu) + int btn_idx = find_button(px, py); + if (btn_idx >= 0) { + auto& btn = buttons_[btn_idx]; + if (btn.id == 99) { // MENU toggle button + drawer_open_ = !drawer_open_; + layout_buttons(); + return true; + } + btn.pressed = true; + btn.finger_id = finger_id; + trigger_button(btn.dik, true); + return true; + } + + // 2. Intelligent UI Hit-Testing: check if touch lands inside an open in-game UI window +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (PythonBoot::IsPointInsideActiveUI(int(px), int(py))) { + ui_touch_finger_id_ = finger_id; + PythonBoot::UIMouseMove(int(px), int(py)); + PythonBoot::UIMouseButton(1, true, int(px), int(py)); + return true; + } +#endif + + // 3. Left side: virtual joystick (lower-left quadrant) + if (norm_x < 0.40f && norm_y > 0.35f && !joystick_active_) { + joystick_active_ = true; + joystick_finger_id_ = finger_id; + joystick_base_x_ = px; + joystick_base_y_ = py; + joystick_knob_x_ = px; + joystick_knob_y_ = py; + update_joystick_motion(px, py); + return true; + } + + // 4. 3D Game World: camera rotation, pinch zoom, or tap to target + if (camera_finger_id_ < 0) { + camera_finger_id_ = finger_id; + camera_last_x_ = px; + camera_last_y_ = py; + camera_start_x_ = px; + camera_start_y_ = py; + camera_dragged_ = false; + camera_down_time_ = get_time_sec(); + return true; + } else if (pinch_finger2_ < 0) { + pinch_finger1_ = camera_finger_id_; + pinch_finger2_ = finger_id; + pinch_last_dist_ = std::hypot(px - camera_last_x_, py - camera_last_y_); + return true; + } + + return false; + } + + // Handles finger motion + bool on_finger_motion(int64_t finger_id, float norm_x, float norm_y) { + if (!enabled_) return false; + const float px = norm_x * float(screen_w_); + const float py = norm_y * float(screen_h_); + + // 1. UI Touch dragging (e.g. dragging item in inventory or scrollbar) + if (ui_touch_finger_id_ == finger_id) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::UIMouseMove(int(px), int(py)); +#endif + return true; + } + + // 2. Virtual Joystick finger + if (joystick_active_ && finger_id == joystick_finger_id_) { + update_joystick_motion(px, py); + return true; + } + + // 3. Button drag tracking (check if finger slid off) + for (auto& btn : buttons_) { + if (btn.finger_id == finger_id) { + const float dist = std::hypot(px - btn.x, py - btn.y); + if (dist > btn.radius * 1.5f && btn.pressed) { + btn.pressed = false; + trigger_button(btn.dik, false); + } else if (dist <= btn.radius * 1.5f && !btn.pressed) { + btn.pressed = true; + trigger_button(btn.dik, true); + } + return true; + } + } + + // 4. Two-finger pinch zoom + if (pinch_finger1_ >= 0 && pinch_finger2_ >= 0 && + (finger_id == pinch_finger1_ || finger_id == pinch_finger2_)) { + const float cur_dist = std::hypot(px - camera_last_x_, py - camera_last_y_); + if (pinch_last_dist_ > 1.0f && cur_dist > 1.0f) { + const float delta_d = cur_dist - pinch_last_dist_; + if (std::abs(delta_d) > 2.0f) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::UIMouseWheel(int(delta_d * 8.0f)); +#endif + pinch_last_dist_ = cur_dist; + } + } + return true; + } + + // 5. Single-finger camera drag + if (finger_id == camera_finger_id_) { + const float total_dist = std::hypot(px - camera_start_x_, py - camera_start_y_); + if (total_dist > 6.0f) { + if (!camera_dragged_) { + camera_dragged_ = true; +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::CameraBeginDrag(int(camera_start_x_), int(camera_start_y_)); +#endif + } +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::CameraDrag(int(px), int(py)); +#endif + } + camera_last_x_ = px; + camera_last_y_ = py; + return true; + } + + return false; + } + + // Handles finger touch up + bool on_finger_up(int64_t finger_id, float norm_x, float norm_y) { + if (!enabled_) return false; + const float px = norm_x * float(screen_w_); + const float py = norm_y * float(screen_h_); + + // 1. UI Touch release (e.g. dropped item in inventory or clicked button) + if (ui_touch_finger_id_ == finger_id) { + ui_touch_finger_id_ = -1; +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::UIMouseButton(1, false, int(px), int(py)); +#endif + return true; + } + + // 2. Joystick release + if (joystick_active_ && finger_id == joystick_finger_id_) { + joystick_active_ = false; + joystick_finger_id_ = -1; + joystick_knob_x_ = joystick_base_x_; + joystick_knob_y_ = joystick_base_y_; +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::SetMoveDirection(0.0f, false); +#endif + return true; + } + + // 3. Button release + for (auto& btn : buttons_) { + if (btn.finger_id == finger_id) { + if (btn.pressed) { + btn.pressed = false; + trigger_button(btn.dik, false); + } + btn.finger_id = -1; + return true; + } + } + + // 4. Pinch end + if (finger_id == pinch_finger1_ || finger_id == pinch_finger2_) { + pinch_finger1_ = -1; + pinch_finger2_ = -1; + pinch_last_dist_ = 0.0f; + if (finger_id == camera_finger_id_) camera_finger_id_ = -1; + return true; + } + + // 5. Camera finger release + if (finger_id == camera_finger_id_) { + camera_finger_id_ = -1; + if (camera_dragged_) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::CameraEndDrag(); +#endif + } else if ((get_time_sec() - camera_down_time_) < 0.35) { + // Short tap on 3D world: select target (mob, NPC, ground) +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::UIMouseMove(int(px), int(py)); + PythonBoot::UIMouseButton(1, true, int(px), int(py)); + PythonBoot::UIMouseButton(1, false, int(px), int(py)); +#endif + } + return true; + } + + return false; + } + + // Render virtual joystick, buttons, drawer menu, and player status bar + void append_ui_commands(std::vector& commands) const { + if (!enabled_) return; + + const float W = float(screen_w_); + +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + // --- 1. Top-Left Player Status Bar (Mobile HUD) --- + const auto status = PythonBoot::GetPlayerStatusInfo(); + if (status.max_hp > 0) { + const float bar_x = 20.0f; + const float bar_y = 16.0f; + const float bar_w = 205.0f; + const float bar_h = 44.0f; + + // Background plate + draw_rect_bar(commands, bar_x, bar_y, bar_x + bar_w, bar_y + bar_h, 0x90101824); + draw_rect_lines(commands, bar_x, bar_y, bar_x + bar_w, bar_y + bar_h, 0x60405870); + + // Avatar circle badge + const float av_cx = bar_x + 22.0f; + const float av_cy = bar_y + 22.0f; + const float av_r = 15.0f; + draw_filled_disc(commands, av_cx, av_cy, av_r, 0xB0203040); + draw_circle(commands, av_cx, av_cy, av_r, 0xE0FFD700, 16); + // "Lv" tick inside avatar + draw_line(commands, av_cx - 4.0f, av_cy - 4.0f, av_cx - 4.0f, av_cy + 3.0f, 0xFFFFFFFF); + draw_line(commands, av_cx - 4.0f, av_cy + 3.0f, av_cx - 1.0f, av_cy + 3.0f, 0xFFFFFFFF); + draw_line(commands, av_cx + 1.0f, av_cy - 4.0f, av_cx + 3.0f, av_cy + 3.0f, 0xFFFFFFFF); + draw_line(commands, av_cx + 5.0f, av_cy - 4.0f, av_cx + 3.0f, av_cy + 3.0f, 0xFFFFFFFF); + + // Gauges + const float gx1 = bar_x + 44.0f; + const float gx2 = bar_x + bar_w - 10.0f; + const float gw = gx2 - gx1; + + // HP Gauge + const float hp_y1 = bar_y + 10.0f; + const float hp_y2 = hp_y1 + 10.0f; + const float hp_ratio = std::clamp(float(status.hp) / float(status.max_hp), 0.0f, 1.0f); + draw_rect_bar(commands, gx1, hp_y1, gx2, hp_y2, 0x70301010); + draw_rect_bar(commands, gx1, hp_y1, gx1 + gw * hp_ratio, hp_y2, 0xE5D82424); + draw_line(commands, gx1, hp_y1, gx1 + gw * hp_ratio, hp_y1, 0x60FFFFFF); // Gloss line + draw_rect_lines(commands, gx1, hp_y1, gx2, hp_y2, 0x80502020); + + // MP Gauge + const float mp_y1 = bar_y + 24.0f; + const float mp_y2 = mp_y1 + 8.0f; + const float mp_ratio = status.max_sp > 0 ? std::clamp(float(status.sp) / float(status.max_sp), 0.0f, 1.0f) : 0.0f; + draw_rect_bar(commands, gx1, mp_y1, gx2, mp_y2, 0x70102038); + draw_rect_bar(commands, gx1, mp_y1, gx1 + gw * mp_ratio, mp_y2, 0xE52068E0); + draw_line(commands, gx1, mp_y1, gx1 + gw * mp_ratio, mp_y1, 0x60FFFFFF); // Gloss line + draw_rect_lines(commands, gx1, mp_y1, gx2, mp_y2, 0x80204060); + + // EXP bar (bottom trim line) + if (status.max_exp > 0) { + const float exp_ratio = std::clamp(float(status.exp) / float(status.max_exp), 0.0f, 1.0f); + draw_rect_bar(commands, bar_x, bar_y + bar_h - 2.0f, bar_x + bar_w * exp_ratio, bar_y + bar_h, 0xD0E0B020); + } + } +#endif + + // --- 2. Top-Right Drawer Menu Bar --- + if (drawer_open_) { + // Background capsule tray behind drawer buttons + draw_rect_bar(commands, W - 468.0f, 16.0f, W - 198.0f, 54.0f, 0x90101824); + draw_rect_lines(commands, W - 468.0f, 16.0f, W - 198.0f, 54.0f, 0x60506880); + } + + // --- 3. Buttons Rendering (Combat wheel & Drawer items) --- + for (const auto& btn : buttons_) { + if (btn.id >= 10 && btn.id <= 15 && !drawer_open_) + continue; + + const uint32_t col = btn.pressed ? btn.color_pressed : btn.color_idle; + draw_filled_disc(commands, btn.x, btn.y, btn.radius, col); + draw_circle(commands, btn.x, btn.y, btn.radius * 0.85f, btn.pressed ? 0xFFFFFFFF : 0x70FFFFFF, 16); + + const uint32_t icon_col = btn.pressed ? 0xFFFFFFFF : 0xDDFFFFFF; + const float r = btn.radius; + + switch (btn.id) { + case 1: { // ATK: crossed swords + const float s = r * 0.35f; + draw_line(commands, btn.x - s, btn.y - s, btn.x + s, btn.y + s, icon_col); + draw_line(commands, btn.x + s, btn.y - s, btn.x - s, btn.y + s, icon_col); + const float g = s * 0.35f; + draw_line(commands, btn.x - s*0.4f - g, btn.y - s*0.4f + g, btn.x - s*0.4f + g, btn.y - s*0.4f - g, icon_col); + draw_line(commands, btn.x + s*0.4f - g, btn.y - s*0.4f - g, btn.x + s*0.4f + g, btn.y - s*0.4f + g, icon_col); + break; + } + case 2: { // S1: I + const float h = r * 0.35f; + draw_line(commands, btn.x, btn.y - h, btn.x, btn.y + h, icon_col); + draw_line(commands, btn.x - 4.0f, btn.y - h, btn.x + 4.0f, btn.y - h, icon_col); + draw_line(commands, btn.x - 4.0f, btn.y + h, btn.x + 4.0f, btn.y + h, icon_col); + break; + } + case 3: { // S2: II + const float h = r * 0.35f; + draw_line(commands, btn.x - 4.0f, btn.y - h, btn.x - 4.0f, btn.y + h, icon_col); + draw_line(commands, btn.x + 4.0f, btn.y - h, btn.x + 4.0f, btn.y + h, icon_col); + break; + } + case 4: { // S3: III + const float h = r * 0.35f; + draw_line(commands, btn.x - 6.0f, btn.y - h, btn.x - 6.0f, btn.y + h, icon_col); + draw_line(commands, btn.x, btn.y - h, btn.x, btn.y + h, icon_col); + draw_line(commands, btn.x + 6.0f, btn.y - h, btn.x + 6.0f, btn.y + h, icon_col); + break; + } + case 5: { // POT: + + const float p = r * 0.4f; + draw_line(commands, btn.x - p, btn.y, btn.x + p, btn.y, icon_col); + draw_line(commands, btn.x, btn.y - p, btn.x, btn.y + p, icon_col); + break; + } + case 6: { // PICK: Downward arrow + const float a = r * 0.35f; + draw_line(commands, btn.x - a, btn.y - a * 0.3f, btn.x, btn.y + a * 0.6f, icon_col); + draw_line(commands, btn.x + a, btn.y - a * 0.3f, btn.x, btn.y + a * 0.6f, icon_col); + draw_line(commands, btn.x, btn.y - a * 0.7f, btn.x, btn.y + a * 0.6f, icon_col); + break; + } + case 99: { // MENU: ☰ hamburger icon + draw_line(commands, btn.x - 7.0f, btn.y - 5.0f, btn.x + 7.0f, btn.y - 5.0f, icon_col); + draw_line(commands, btn.x - 7.0f, btn.y, btn.x + 7.0f, btn.y, icon_col); + draw_line(commands, btn.x - 7.0f, btn.y + 5.0f, btn.x + 7.0f, btn.y + 5.0f, icon_col); + break; + } + case 10: { // BAG: Backpack + draw_rect_lines(commands, btn.x - 6.0f, btn.y - 4.0f, btn.x + 6.0f, btn.y + 6.0f, icon_col); + draw_line(commands, btn.x - 3.0f, btn.y - 4.0f, btn.x, btn.y - 7.0f, icon_col); + draw_line(commands, btn.x, btn.y - 7.0f, btn.x + 3.0f, btn.y - 4.0f, icon_col); + draw_line(commands, btn.x - 6.0f, btn.y, btn.x + 6.0f, btn.y, icon_col); + break; + } + case 11: { // CHAR: Head + Shoulders + draw_circle(commands, btn.x, btn.y - 3.0f, 4.0f, icon_col, 12); + draw_line(commands, btn.x - 6.0f, btn.y + 6.0f, btn.x - 3.0f, btn.y + 2.0f, icon_col); + draw_line(commands, btn.x - 3.0f, btn.y + 2.0f, btn.x + 3.0f, btn.y + 2.0f, icon_col); + draw_line(commands, btn.x + 3.0f, btn.y + 2.0f, btn.x + 6.0f, btn.y + 6.0f, icon_col); + break; + } + case 12: { // SKILL: Lightning + draw_line(commands, btn.x + 2.0f, btn.y - 7.0f, btn.x - 3.0f, btn.y - 1.0f, icon_col); + draw_line(commands, btn.x - 3.0f, btn.y - 1.0f, btn.x + 1.0f, btn.y - 1.0f, icon_col); + draw_line(commands, btn.x + 1.0f, btn.y - 1.0f, btn.x - 2.0f, btn.y + 7.0f, icon_col); + break; + } + case 13: { // QUEST: Scroll + draw_rect_lines(commands, btn.x - 5.0f, btn.y - 6.0f, btn.x + 5.0f, btn.y + 6.0f, icon_col); + draw_line(commands, btn.x - 3.0f, btn.y - 2.0f, btn.x + 3.0f, btn.y - 2.0f, icon_col); + draw_line(commands, btn.x - 3.0f, btn.y + 2.0f, btn.x + 1.0f, btn.y + 2.0f, icon_col); + break; + } + case 14: { // COMM: Chat bubble + draw_rect_lines(commands, btn.x - 6.0f, btn.y - 5.0f, btn.x + 6.0f, btn.y + 3.0f, icon_col); + draw_line(commands, btn.x - 3.0f, btn.y + 3.0f, btn.x - 5.0f, btn.y + 6.0f, icon_col); + draw_line(commands, btn.x - 5.0f, btn.y + 6.0f, btn.x, btn.y + 3.0f, icon_col); + break; + } + case 15: { // SET: Gear / Close + draw_circle(commands, btn.x, btn.y, 4.0f, icon_col, 10); + draw_line(commands, btn.x - 7.0f, btn.y, btn.x + 7.0f, btn.y, icon_col); + draw_line(commands, btn.x, btn.y - 7.0f, btn.x, btn.y + 7.0f, icon_col); + draw_line(commands, btn.x - 5.0f, btn.y - 5.0f, btn.x + 5.0f, btn.y + 5.0f, icon_col); + draw_line(commands, btn.x - 5.0f, btn.y + 5.0f, btn.x + 5.0f, btn.y - 5.0f, icon_col); + break; + } + } + } + + // --- 4. Virtual Joystick --- + draw_circle(commands, joystick_base_x_, joystick_base_y_, joystick_radius_, 0x8080C0FF, 24); + draw_circle(commands, joystick_base_x_, joystick_base_y_, joystick_radius_ * 0.45f, 0x4080C0FF, 16); + draw_line(commands, joystick_base_x_ - joystick_radius_, joystick_base_y_, + joystick_base_x_ - joystick_radius_ + 8.0f, joystick_base_y_, 0x90FFFFFF); + draw_line(commands, joystick_base_x_ + joystick_radius_ - 8.0f, joystick_base_y_, + joystick_base_x_ + joystick_radius_, joystick_base_y_, 0x90FFFFFF); + draw_line(commands, joystick_base_x_, joystick_base_y_ - joystick_radius_, + joystick_base_x_, joystick_base_y_ - joystick_radius_ + 8.0f, 0x90FFFFFF); + draw_line(commands, joystick_base_x_, joystick_base_y_ + joystick_radius_ - 8.0f, + joystick_base_x_, joystick_base_y_ + joystick_radius_, 0x90FFFFFF); + + if (joystick_active_) { + draw_line(commands, joystick_base_x_, joystick_base_y_, joystick_knob_x_, joystick_knob_y_, 0xB000FFFF); + } + const uint32_t knob_color = joystick_active_ ? 0xB040A0FF : 0x6040A0FF; + draw_filled_disc(commands, joystick_knob_x_, joystick_knob_y_, joystick_knob_radius_, knob_color); + draw_circle(commands, joystick_knob_x_, joystick_knob_y_, joystick_knob_radius_ * 0.5f, 0x80FFFFFF, 12); + } + + // Desktop mouse testing simulation + bool on_mouse_button(int button, bool pressed, int x, int y) { + if (!enabled_) return false; + const float norm_x = float(x) / float(screen_w_); + const float norm_y = float(y) / float(screen_h_); + if (pressed) { + if (button == 1) { + return on_finger_down(101, norm_x, norm_y); + } + return false; + } else { + if (button == 1) { + return on_finger_up(101, norm_x, norm_y); + } + return false; + } + } + + bool on_mouse_motion(int x, int y) { + if (!enabled_) return false; + const float norm_x = float(x) / float(screen_w_); + const float norm_y = float(y) / float(screen_h_); + bool handled = false; + if (ui_touch_finger_id_ == 101) { + handled |= on_finger_motion(101, norm_x, norm_y); + } + if (joystick_active_ && joystick_finger_id_ == 101) { + handled |= on_finger_motion(101, norm_x, norm_y); + } + if (camera_finger_id_ == 101) { + handled |= on_finger_motion(101, norm_x, norm_y); + } + return handled; + } + +private: + void init_buttons() { + buttons_.clear(); + // Combat Action Wheel (Lower-Right) + buttons_.push_back({1, 0x39, 0, 0, 42.0f, "ATK", 0x80D48820, 0xD0FFB040}); + buttons_.push_back({2, 0x02, 0, 0, 26.0f, "S1", 0x803060C0, 0xD05080FF}); + buttons_.push_back({3, 0x03, 0, 0, 26.0f, "S2", 0x80903090, 0xD0D050D0}); + buttons_.push_back({4, 0x04, 0, 0, 26.0f, "S3", 0x80309060, 0xD050D080}); + buttons_.push_back({5, 0x05, 0, 0, 22.0f, "POT", 0x90A03030, 0xD0FF5050}); + buttons_.push_back({6, 0x2c, 0, 0, 22.0f, "PICK", 0x80208080, 0xD040B0B0}); + + // Top-Right Drawer Menu Toggle + buttons_.push_back({99, 0, 0, 0, 20.0f, "MENU", 0x90283848, 0xD0FFB040}); + + // Drawer Menu Buttons + buttons_.push_back({10, 0x17, 0, 0, 18.0f, "BAG", 0x90D09020, 0xD0FFB040}); // DIK_I + buttons_.push_back({11, 0x2e, 0, 0, 18.0f, "CHAR", 0x903060B0, 0xD05080FF}); // DIK_C + buttons_.push_back({12, 0x2f, 0, 0, 18.0f, "SKILL", 0x90803090, 0xD0D050D0}); // DIK_V + buttons_.push_back({13, 0x31, 0, 0, 18.0f, "QUEST", 0x90308050, 0xD050D070}); // DIK_N + buttons_.push_back({14, 0x32, 0, 0, 18.0f, "COMM", 0x90905020, 0xD0E07030}); // DIK_M + buttons_.push_back({15, 0x01, 0, 0, 18.0f, "SET", 0x90506070, 0xD08090A0}); // DIK_ESCAPE + } + + void layout_buttons() { + const float W = float(screen_w_); + const float H = float(screen_h_); + for (auto& btn : buttons_) { + switch (btn.id) { + case 1: + btn.x = W - 85.0f; + btn.y = H - 85.0f; + btn.radius = std::min(46.0f, H * 0.12f); + break; + case 2: + btn.x = W - 165.0f; + btn.y = H - 85.0f; + btn.radius = std::min(28.0f, H * 0.08f); + break; + case 3: + btn.x = W - 145.0f; + btn.y = H - 155.0f; + btn.radius = std::min(28.0f, H * 0.08f); + break; + case 4: + btn.x = W - 85.0f; + btn.y = H - 175.0f; + btn.radius = std::min(28.0f, H * 0.08f); + break; + case 5: + btn.x = W - 225.0f; + btn.y = H - 75.0f; + btn.radius = std::min(24.0f, H * 0.065f); + break; + case 6: + btn.x = W - 85.0f; + btn.y = H - 235.0f; + btn.radius = std::min(24.0f, H * 0.065f); + break; + case 99: // MENU toggle button (to the left of MiniMap) + btn.x = W - 170.0f; + btn.y = 35.0f; + btn.radius = 20.0f; + break; + case 10: // BAG + btn.x = W - 220.0f; + btn.y = 35.0f; + btn.radius = 18.0f; + break; + case 11: // CHAR + btn.x = W - 265.0f; + btn.y = 35.0f; + btn.radius = 18.0f; + break; + case 12: // SKILL + btn.x = W - 310.0f; + btn.y = 35.0f; + btn.radius = 18.0f; + break; + case 13: // QUEST + btn.x = W - 355.0f; + btn.y = 35.0f; + btn.radius = 18.0f; + break; + case 14: // COMM + btn.x = W - 400.0f; + btn.y = 35.0f; + btn.radius = 18.0f; + break; + case 15: // SET + btn.x = W - 445.0f; + btn.y = 35.0f; + btn.radius = 18.0f; + break; + } + } + } + + int find_button(float x, float y) { + for (size_t i = 0; i < buttons_.size(); ++i) { + const auto& btn = buttons_[i]; + if (btn.id >= 10 && btn.id <= 15 && !drawer_open_) + continue; + const float dist = std::hypot(x - btn.x, y - btn.y); + if (dist <= btn.radius * 1.25f) { + return int(i); + } + } + return -1; + } + + void trigger_button(int dik, bool pressed) { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + if (dik == 0x39) { + PythonBoot::SetAttackKey(pressed); + } + if (dik != 0) { + PythonBoot::UIKey(dik, pressed); + } +#endif + } + + void update_joystick_motion(float px, float py) { + const float dx = px - joystick_base_x_; + const float dy = py - joystick_base_y_; + const float dist = std::hypot(dx, dy); + + if (dist <= joystick_radius_) { + joystick_knob_x_ = px; + joystick_knob_y_ = py; + } else if (dist > 0.0f) { + joystick_knob_x_ = joystick_base_x_ + (dx / dist) * joystick_radius_; + joystick_knob_y_ = joystick_base_y_ + (dy / dist) * joystick_radius_; + } + + if (dist > 8.0f) { + const float rad = std::atan2(-dx, -dy); + move_angle_ = rad * 180.0f / 3.14159265358979323846f; + if (move_angle_ < 0.0f) move_angle_ += 360.0f; +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::SetMoveDirection(move_angle_, true); +#endif + } else { +#ifdef MT_NATIVE_HAS_LIVE_CLIENT + PythonBoot::SetMoveDirection(0.0f, false); +#endif + } + } + + static void draw_line(std::vector& commands, + float x1, float y1, float x2, float y2, uint32_t argb) { + UIRenderCommand cmd{}; + cmd.kind = UIRenderCommand::Line; + cmd.x1 = x1; cmd.y1 = y1; + cmd.x2 = x2; cmd.y2 = y2; + cmd.argb = argb; + commands.push_back(cmd); + } + + static void draw_rect_bar(std::vector& commands, + float x1, float y1, float x2, float y2, uint32_t argb) { + UIRenderCommand bar{}; + bar.kind = UIRenderCommand::Bar; + bar.x1 = x1; bar.y1 = y1; + bar.x2 = x2; bar.y2 = y2; + bar.argb = argb; + commands.push_back(bar); + } + + static void draw_rect_lines(std::vector& commands, + float x1, float y1, float x2, float y2, uint32_t argb) { + draw_line(commands, x1, y1, x2, y1, argb); + draw_line(commands, x2, y1, x2, y2, argb); + draw_line(commands, x2, y2, x1, y2, argb); + draw_line(commands, x1, y2, x1, y1, argb); + } + + static void draw_circle(std::vector& commands, + float cx, float cy, float radius, uint32_t argb, int segments = 16) { + const float step = 2.0f * 3.14159265f / float(segments); + for (int i = 0; i < segments; ++i) { + const float a1 = float(i) * step; + const float a2 = float(i + 1) * step; + draw_line(commands, + cx + std::cos(a1) * radius, cy + std::sin(a1) * radius, + cx + std::cos(a2) * radius, cy + std::sin(a2) * radius, + argb); + } + } + + static void draw_filled_disc(std::vector& commands, + float cx, float cy, float radius, uint32_t argb) { + // Base rectangular fill + draw_rect_bar(commands, cx - radius * 0.65f, cy - radius * 0.65f, cx + radius * 0.65f, cy + radius * 0.65f, (argb & 0x00FFFFFF) | 0x55000000); + // Cross fills for roundness + draw_rect_bar(commands, cx - radius * 0.85f, cy - radius * 0.35f, cx + radius * 0.85f, cy + radius * 0.35f, (argb & 0x00FFFFFF) | 0x55000000); + draw_rect_bar(commands, cx - radius * 0.35f, cy - radius * 0.85f, cx + radius * 0.35f, cy + radius * 0.85f, (argb & 0x00FFFFFF) | 0x55000000); + // Border rings + draw_circle(commands, cx, cy, radius, argb, 20); + draw_circle(commands, cx, cy, radius - 1.0f, (argb & 0x00FFFFFF) | 0x40000000, 20); + } + + static double get_time_sec() { + using namespace std::chrono; + return duration_cast>(steady_clock::now().time_since_epoch()).count(); + } + + bool enabled_ = false; + int screen_w_ = 1280; + int screen_h_ = 720; + + bool drawer_open_ = false; + int64_t ui_touch_finger_id_ = -1; + + bool joystick_active_ = false; + int64_t joystick_finger_id_ = -1; + float joystick_base_x_ = 140.0f; + float joystick_base_y_ = 580.0f; + float joystick_knob_x_ = 140.0f; + float joystick_knob_y_ = 580.0f; + float joystick_radius_ = 65.0f; + float joystick_knob_radius_ = 28.0f; + float move_angle_ = 0.0f; + + int64_t camera_finger_id_ = -1; + float camera_start_x_ = 0.0f; + float camera_start_y_ = 0.0f; + float camera_last_x_ = 0.0f; + float camera_last_y_ = 0.0f; + double camera_down_time_ = 0.0; + bool camera_dragged_ = false; + + int64_t pinch_finger1_ = -1; + int64_t pinch_finger2_ = -1; + float pinch_last_dist_ = 0.0f; + + std::vector buttons_; +}; diff --git a/project/debug_actor_mesh.gd b/project/debug_actor_mesh.gd new file mode 100644 index 00000000..936cf871 --- /dev/null +++ b/project/debug_actor_mesh.gd @@ -0,0 +1,140 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + world = Python3DSurface.new() + world.name = "Python3DSurface" + ui = PythonUISurface.new() + ui.name = "PythonUISurface" + root.add_child(ui) + root.add_child(world) + + var err: String = ui.run_app() + if err != "": + print("ERROR: run_app: ", err) + quit(1) + return + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + var host = OS.get_environment("MT_LIVE_HOST") + var login = OS.get_environment("MT_LIVE_LOGIN") + var pwd = OS.get_environment("MT_LIVE_PASSWORD") + var auth_port = int(OS.get_environment("MT_LIVE_AUTH_PORT")) + var game_port = int(OS.get_environment("MT_LIVE_GAME_PORT")) + + _py("_stream.SetConnectInfo('%s', %d, '%s', %d)\n" % [host, game_port, host, auth_port] + + "_stream.curPhaseWindow.Connect('%s', '%s')" % [login, pwd]) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + _py("_stream.curPhaseWindow.SelectSlot(0)\n_stream.curPhaseWindow.StartGame()") + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()" + )) + + for i in 60: + await process_frame + + print("=== CAMERA WORLD & HEIGHT PROFILE ===") + var draws: Array = Metin2PythonHost.render3d_draws() + if not draws.is_empty(): + var vm: PackedFloat32Array = draws[0]["view"] + var cam_inv: Transform3D = Python3DSurface.d3d_transform(vm).affine_inverse() + print("Camera world origin: ", cam_inv.origin) + print("Camera world basis: ", cam_inv.basis) + print("Camera looking direction: ", -cam_inv.basis.z) + + for y in range(68200, 66200, -100): + var bg_h = Metin2PythonHost.evaluate("str(background.GetHeight(59700.0, %f))" % float(y)) + var terrain_h = float(world.terrain.call("sample_height", 597.0, float(y) / 100.0)) * 100.0 if world.terrain else 0.0 + print("Profile y=%d: bg_h=%s terrain_h=%.2f" % [y, bg_h, terrain_h]) + + print("Total draws: ", draws.size()) + + for d in draws: + var tex: String = d.get("texture0", "") + if tex.contains("warrior") or tex.contains("stray_dog") or tex.contains("goods"): + var wm: PackedFloat32Array = d["world"] + var vm: PackedFloat32Array = d["view"] + var xform: Transform3D = Python3DSurface.d3d_transform(wm) + var cam_xform: Transform3D = Python3DSurface.d3d_transform(Python3DSurface.multiply(wm, vm)) + + var min_z := 1e9 + var max_z := -1e9 + var min_local_z := 1e9 + var max_local_z := -1e9 + for pos in d["positions"]: + min_local_z = minf(min_local_z, pos.z) + max_local_z = maxf(max_local_z, pos.z) + var wp: Vector3 = xform * pos + min_z = minf(min_z, wp.z) + max_z = maxf(max_z, wp.z) + + # Map position: (xform.origin.x, -xform.origin.y) + var map_x := xform.origin.x + var map_y := -xform.origin.y + var actor_z := xform.origin.z + + var bg_h = Metin2PythonHost.evaluate("str(background.GetHeight(%f, %f))" % [map_x, map_y]) + var terrain_h = float(world.terrain.call("sample_height", map_x / 100.0, map_y / 100.0)) * 100.0 if world.terrain else 0.0 + + print("Actor [%s]:" % tex.get_file()) + print(" World origin: ", xform.origin, " (map_x=%.1f, map_y=%.1f, origin_z=%.1f)" % [map_x, map_y, actor_z]) + print(" Local vert Z range: [%.2f, %.2f]" % [min_local_z, max_local_z]) + print(" World vert Z range: [%.2f, %.2f]" % [min_z, max_z]) + print(" Background Height: %s, Terrain Height: %.2f" % [bg_h, terrain_h]) + print(" Vert_Z min vs Height: diff = %.2f cm" % (min_z - float(bg_h))) + print(" Origin_Z vs Height: diff = %.2f cm" % (actor_z - float(bg_h))) + print(" Camera-space origin: ", cam_xform.origin) + + if tex.contains("stray_dog"): + print("--- DOG TERRAIN DETAIL ---") + var from_map := Transform3D(Basis(Vector3(100, 0, 0), Vector3(0, 0, 100), Vector3(0, -100, 0)), Vector3.ZERO) + var dog_world := xform.origin # (59700, -66400, 19851.5) + print("Dog 40250 world: ", dog_world) + for child in world.terrain.find_children("*", "MeshInstance3D", true, false): + var mi := child as MeshInstance3D + if mi.mesh is ArrayMesh: + var am := mi.mesh as ArrayMesh + var arrays := am.surface_get_arrays(0) + var verts: PackedVector3Array = arrays[Mesh.ARRAY_VERTEX] + for v in verts: + var world_v: Vector3 = from_map * (mi.transform * v) + if abs(world_v.x - dog_world.x) < 300.0 and abs(world_v.y - dog_world.y) < 300.0: + print("Terrain local vert: ", v, " -> mi.xform*v: ", mi.transform * v, " -> from_map: ", world_v) + print("Diff Z (terrain world Z - dog world Z): ", world_v.z - dog_world.z) + break + + + quit(0) diff --git a/project/debug_actor_mesh.gd.uid b/project/debug_actor_mesh.gd.uid new file mode 100644 index 00000000..4b4ad01a --- /dev/null +++ b/project/debug_actor_mesh.gd.uid @@ -0,0 +1 @@ +uid://b5vx3h3lipsuf diff --git a/project/debug_anomalies.gd b/project/debug_anomalies.gd new file mode 100644 index 00000000..859cb7ad --- /dev/null +++ b/project/debug_anomalies.gd @@ -0,0 +1,116 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + world = Python3DSurface.new() + ui = PythonUISurface.new() + root.add_child(ui) + root.add_child(world) + + ui.run_app() + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + var host = OS.get_environment("MT_LIVE_HOST") + var login = OS.get_environment("MT_LIVE_LOGIN") + var pwd = OS.get_environment("MT_LIVE_PASSWORD") + var auth_port = int(OS.get_environment("MT_LIVE_AUTH_PORT")) + var game_port = int(OS.get_environment("MT_LIVE_GAME_PORT")) + + _py("_stream.SetConnectInfo('%s', %d, '%s', %d)\n" % [host, game_port, host, auth_port] + + "_stream.curPhaseWindow.Connect('%s', '%s')" % [login, pwd]) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + _py("_stream.curPhaseWindow.SelectSlot(0)\n_stream.curPhaseWindow.StartGame()") + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()" + )) + + for i in 60: + await process_frame + + print("=== INVESTIGATING ALL 518 ACTORS ===") + var draws: Array = Metin2PythonHost.render3d_draws() + var actor_draws := {} + for d in draws: + var wm: PackedFloat32Array = d["world"] + var key := "%.1f,%.1f,%.1f" % [wm[12], wm[13], wm[14]] + if not actor_draws.has(key): + actor_draws[key] = [] + actor_draws[key].append(d) + + print("Unique actor positions: ", actor_draws.size()) + + var anomalies := [] + for key in actor_draws: + var ds: Array = actor_draws[key] + var wm: PackedFloat32Array = ds[0]["world"] + var map_x := wm[12] + var map_y := -wm[13] + var actor_z := wm[14] + + var bg_h = float(Metin2PythonHost.evaluate("str(background.GetHeight(%f, %f))" % [map_x, map_y])) + var terrain_h = float(world.terrain.call("sample_height", map_x / 100.0, map_y / 100.0)) * 100.0 if world.terrain else 0.0 + + var diff_bg = actor_z - bg_h + var diff_terrain = actor_z - terrain_h + + # Find textures + var texs := [] + var min_vert_z := 1e9 + for d in ds: + var t: String = d.get("texture0", "") + if not t.is_empty(): + texs.append(t.get_file()) + for p in d["positions"]: + min_vert_z = minf(min_vert_z, actor_z + p.z) + + if absf(diff_bg) > 5.0 or absf(diff_terrain) > 5.0: + anomalies.append({ + "pos": [map_x, map_y, actor_z], + "bg_h": bg_h, + "terrain_h": terrain_h, + "diff_bg": diff_bg, + "diff_terrain": diff_terrain, + "vert_min_diff": min_vert_z - bg_h, + "textures": texs + }) + + print("Found %d anomalies with gap > 5cm!" % anomalies.size()) + for a in anomalies.slice(0, 30): + print("Anomaly: pos=(%.1f, %.1f, %.1f), bg_h=%.1f, terrain_h=%.1f, diff_bg=%.1f, diff_terrain=%.1f, vert_min_diff=%.1f, tex=%s" % [ + a["pos"][0], a["pos"][1], a["pos"][2], a["bg_h"], a["terrain_h"], a["diff_bg"], a["diff_terrain"], a["vert_min_diff"], a["textures"] + ]) + + quit(0) diff --git a/project/debug_anomalies.gd.uid b/project/debug_anomalies.gd.uid new file mode 100644 index 00000000..ff597280 --- /dev/null +++ b/project/debug_anomalies.gd.uid @@ -0,0 +1 @@ +uid://xn0yhjq7h8d diff --git a/project/debug_height.gd b/project/debug_height.gd new file mode 100644 index 00000000..cfcbc4bb --- /dev/null +++ b/project/debug_height.gd @@ -0,0 +1,143 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _key(keycode: Key, unicode: int, pressed: bool) -> void: + var event := InputEventKey.new() + event.keycode = keycode + event.physical_keycode = keycode + event.unicode = unicode + event.pressed = pressed + root.push_input(event) + +func _press(keycode: Key) -> void: + _key(keycode, 0, true) + _key(keycode, 0, false) + +func _type(text: String) -> void: + for i in text.length(): + var ch := text.unicode_at(i) + var keycode := OS.find_keycode_from_string(text[i].to_upper()) + _key(keycode, ch, true) + _key(keycode, ch, false) + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + + world = Python3DSurface.new() + world.name = "Python3DSurface" + ui = PythonUISurface.new() + ui.name = "PythonUISurface" + root.add_child(ui) + root.add_child(world) + + var err: String = ui.run_app() + if err != "": + print("ERROR: run_app failed: ", err) + quit(1) + return + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + _py( + "_env = os.environ\n" + + "_slot = int(_env.get('MT_LIVE_SLOT', '0'))\n" + + "_stream.SetConnectInfo(_env['MT_LIVE_HOST'], int(_env.get('MT_LIVE_GAME_PORT', '13000')), " + + "_env['MT_LIVE_HOST'], int(_env.get('MT_LIVE_AUTH_PORT', '11000')))\n" + + "_w = _stream.curPhaseWindow\n" + + "_w._LoginWindow__OpenLoginBoard()\n" + + "_w.idEditLine.SetText('')\n" + + "_w.pwdEditLine.SetText('')\n" + + "_w.idEditLine.SetFocus()" + ) + _type(OS.get_environment("MT_LIVE_LOGIN")) + _press(KEY_TAB) + _type(OS.get_environment("MT_LIVE_PASSWORD")) + _py("del _env, _w") + _press(KEY_ENTER) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + _py("_stream.curPhaseWindow.SelectSlot(_slot)\n_stream.curPhaseWindow.StartGame()") + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()" + )) + + for i in 30: + await process_frame + + print("=== DEBUG HEIGHT REPORT ===") + var test_points = [ + Vector2(59700, 68200), + Vector2(59700, 66400), + Vector2(60100, 66400), + Vector2(50000, 50000), + Vector2(30000, 30000), + Vector2(70000, 70000), + ] + for pt in test_points: + var bg_h = Metin2PythonHost.evaluate("str(background.GetHeight(%f, %f))" % [pt.x, pt.y]) + var tm_h = float(world.terrain.call("sample_height", pt.x / 100.0, pt.y / 100.0)) * 100.0 if world.terrain else 0.0 + print("Point (%.0f, %.0f): 40250_bg_h=%s, Metin2World_h=%.1f, diff=%.1f" % [ + pt.x, pt.y, bg_h, tm_h, float(bg_h) - tm_h + ]) + + # Inspect terrain mesh triangles around dog + print("=== TERRAIN MESH VERTICES AROUND DOG (59700, -66400) ===") + # In camera space (since Camera3D is at origin): + # Dog world pos in 40250: (59700, -66400, 19851.5) + # What is dog in camera space? + var draws = Metin2PythonHost.render3d_draws() + for i in draws.size(): + var d = draws[i] + var tex: String = d.get("texture0", "") + if tex.contains("stray_dog"): + var wm: PackedFloat32Array = d.get("world", PackedFloat32Array()) + var vm: PackedFloat32Array = d.get("view", PackedFloat32Array()) + var xform: Transform3D = Python3DSurface.d3d_transform(Python3DSurface.multiply(wm, vm)) + print("DOG camera-space transform origin: ", xform.origin) + + # Now let's check terrain mesh global transform and vertices: + for child in world.terrain.find_children("*", "MeshInstance3D", true, false): + var mi := child as MeshInstance3D + if mi.mesh is ArrayMesh: + var am := mi.mesh as ArrayMesh + var arrays := am.surface_get_arrays(0) + var verts: PackedVector3Array = arrays[Mesh.ARRAY_VERTEX] + for v in verts: + var gv: Vector3 = mi.global_transform * v + # Dog in camera space is around xform.origin. Let's find terrain vertices within 200cm: + if abs(gv.x - xform.origin.x) < 200.0 and abs(gv.z - xform.origin.z) < 200.0: + print("Terrain vertex near dog in camera-space: ", gv, " dog origin: ", xform.origin, " diff_y (height in cam space): ", gv.y - xform.origin.y) + break + break + + quit(0) diff --git a/project/debug_height.gd.uid b/project/debug_height.gd.uid new file mode 100644 index 00000000..ffc89dd1 --- /dev/null +++ b/project/debug_height.gd.uid @@ -0,0 +1 @@ +uid://mpcddwfiipkc diff --git a/project/debug_select_screen.gd b/project/debug_select_screen.gd new file mode 100644 index 00000000..d14d9c3b --- /dev/null +++ b/project/debug_select_screen.gd @@ -0,0 +1,71 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + world = Python3DSurface.new() + world.name = "Python3DSurface" + ui = PythonUISurface.new() + ui.name = "PythonUISurface" + root.add_child(ui) + root.add_child(world) + + var err: String = ui.run_app() + if err != "": + print("ERROR: run_app: ", err) + quit(1) + return + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + var host = OS.get_environment("MT_LIVE_HOST") + var login = OS.get_environment("MT_LIVE_LOGIN") + var pwd = OS.get_environment("MT_LIVE_PASSWORD") + var auth_port = int(OS.get_environment("MT_LIVE_AUTH_PORT")) + var game_port = int(OS.get_environment("MT_LIVE_GAME_PORT")) + + _py("_stream.SetConnectInfo('%s', %d, '%s', %d)\n" % [host, game_port, host, auth_port] + + "_stream.curPhaseWindow.Connect('%s', '%s')" % [login, pwd]) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + print("=== IN INTRO SELECT PHASE ===") + for i in 60: + await process_frame + + var cmds: Array = Metin2PythonHost.ui_render_commands() + print("UI commands count: ", cmds.size()) + var file = FileAccess.open("/tmp/select_ui_cmds.json", FileAccess.WRITE) + file.store_string(JSON.stringify(cmds, " ")) + file.close() + print("Saved /tmp/select_ui_cmds.json") + + quit(0) diff --git a/project/debug_select_screen.gd.uid b/project/debug_select_screen.gd.uid new file mode 100644 index 00000000..c511e1c0 --- /dev/null +++ b/project/debug_select_screen.gd.uid @@ -0,0 +1 @@ +uid://c0k854rmu55u7 diff --git a/project/debug_stones.gd b/project/debug_stones.gd new file mode 100644 index 00000000..da3aafca --- /dev/null +++ b/project/debug_stones.gd @@ -0,0 +1,70 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + var world = Python3DSurface.new() + var ui = PythonUISurface.new() + root.add_child(ui) + root.add_child(world) + + ui.run_app() + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + await _pump_until(20, func(): return _py("assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)")) + + var host = OS.get_environment("MT_LIVE_HOST") + var login = OS.get_environment("MT_LIVE_LOGIN") + var pwd = OS.get_environment("MT_LIVE_PASSWORD") + var auth_port = int(OS.get_environment("MT_LIVE_AUTH_PORT")) + var game_port = int(OS.get_environment("MT_LIVE_GAME_PORT")) + + _py("_stream.SetConnectInfo('%s', %d, '%s', %d)\n_stream.curPhaseWindow.Connect('%s', '%s')" % [host, game_port, host, auth_port, login, pwd]) + await _pump_until(30, func(): return _py("assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)")) + _py("_stream.curPhaseWindow.SelectSlot(0)\n_stream.curPhaseWindow.StartGame()") + await _pump_until(30, func(): return _py("assert isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()")) + + for i in 60: + await process_frame + + print("=== INSPECTING STONE TILES AROUND PLAYER (59700, 68200) ===") + var draws: Array = Metin2PythonHost.render3d_draws() + var player_z := 19851.5 + + for d in draws: + var tex: String = d.get("texture0", "") + if tex.contains("stone") or tex.contains("floor") or tex.contains("tile") or tex.contains("sign"): + var wm: PackedFloat32Array = d["world"] + var xform: Transform3D = Python3DSurface.d3d_transform(wm) + # check if this object is near player (within 2000cm): + if absf(xform.origin.x - 59700.0) < 3000.0 and absf(-xform.origin.y - 68200.0) < 3000.0: + var min_z := 1e9 + var max_z := -1e9 + for p in d["positions"]: + var wp: Vector3 = xform * p + min_z = minf(min_z, wp.z) + max_z = maxf(max_z, wp.z) + print("Object [%s] at (%.1f, %.1f, %.1f), Z range: [%.2f, %.2f], diff vs player_z(19851.5): min=%.2f, max=%.2f" % [ + tex.get_file(), xform.origin.x, -xform.origin.y, xform.origin.z, min_z, max_z, min_z - player_z, max_z - player_z + ]) + + quit(0) diff --git a/project/debug_stones.gd.uid b/project/debug_stones.gd.uid new file mode 100644 index 00000000..035e9be4 --- /dev/null +++ b/project/debug_stones.gd.uid @@ -0,0 +1 @@ +uid://cy6rh17pn00hd diff --git a/project/debug_weapon_motion.gd b/project/debug_weapon_motion.gd new file mode 100644 index 00000000..95caa15a --- /dev/null +++ b/project/debug_weapon_motion.gd @@ -0,0 +1,115 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + world = Python3DSurface.new() + world.name = "Python3DSurface" + ui = PythonUISurface.new() + ui.name = "PythonUISurface" + root.add_child(ui) + root.add_child(world) + + var err: String = ui.run_app() + if err != "": + print("ERROR: run_app: ", err) + quit(1) + return + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr, app\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + var host = OS.get_environment("MT_LIVE_HOST") + var login = OS.get_environment("MT_LIVE_LOGIN") + var pwd = OS.get_environment("MT_LIVE_PASSWORD") + var auth_port = int(OS.get_environment("MT_LIVE_AUTH_PORT")) + var game_port = int(OS.get_environment("MT_LIVE_GAME_PORT")) + + _py("_stream.SetConnectInfo('%s', %d, '%s', %d)\n" % [host, game_port, host, auth_port] + + "_stream.curPhaseWindow.Connect('%s', '%s')" % [login, pwd]) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + _py("_stream.curPhaseWindow.SelectSlot(0)\n_stream.curPhaseWindow.StartGame()") + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()" + )) + + print("=== IN-GAME LOADED ===") + for i in 10: + await process_frame + + print("--- EQUIPPING WEAPON 10 ---") + _py("chr.SelectInstance(player.GetMainCharacterIndex())\nchr.SetWeapon(10)") + for i in 10: + await process_frame + + var check_draws = func(label: String): + var draws: Array = Metin2PythonHost.render3d_draws() + var body_draw = null + var weapon_draws = [] + for d in draws: + var tex: String = d.get("texture0", "").to_lower() + if tex.contains("warrior") and not tex.contains("hair"): + body_draw = d + elif tex.contains("weapon") or tex.contains("sword") or tex.contains("00010") or tex.contains("00040"): + weapon_draws.append(d) + print("[%s] Total draws=%d, weapon_draws=%d" % [label, draws.size(), weapon_draws.size()]) + var b_xf: Transform3D = Transform3D() + if body_draw != null: + var wm: PackedFloat32Array = body_draw["world"] + b_xf = Python3DSurface.d3d_transform(wm) + print(" Body origin: ", b_xf.origin, " tex=", body_draw.get("texture0", "")) + else: + print(" Body draw not found!") + for w in weapon_draws: + var wm: PackedFloat32Array = w["world"] + var xf: Transform3D = Python3DSurface.d3d_transform(wm) + var dist := xf.origin.distance_to(b_xf.origin) if body_draw != null else 0.0 + print(" Weapon origin: ", xf.origin, " dist_to_body=%.2f cm tex=%s" % [dist, w.get("texture0", "")]) + + check_draws.call("IDLE_FRAME_0") + + # Start moving! + print("--- STARTING MOVEMENT (DIK_UP) ---") + _py("player.SetSingleDIKKeyState(app.DIK_UP, True)") + + for frame in range(1, 40): + await process_frame + if frame % 5 == 0: + check_draws.call("MOVING_FRAME_%d" % frame) + + _py("player.SetSingleDIKKeyState(app.DIK_UP, False)") + for frame in range(1, 15): + await process_frame + check_draws.call("STOPPED") + + quit(0) diff --git a/project/python_3d_surface.gd b/project/python_3d_surface.gd index 53f43c84..84a1788d 100644 --- a/project/python_3d_surface.gd +++ b/project/python_3d_surface.gd @@ -12,11 +12,13 @@ # in the recorded D3D camera's coordinate frame, and its .msenv supplies the character light. extends Node3D +const AssetRoot = preload("res://asset_root.gd") const UiAssets = preload("res://ui/ui_assets.gd") # D3DCULL / D3DBLEND / D3DCMPFUNC values used below (D3D8Types.h). const D3DCULL_NONE := 1 const D3DCULL_CW := 2 +const D3DCULL_CCW := 3 const D3DBLEND_ONE := 2 const D3DBLEND_SRCALPHA := 5 const D3DBLEND_INVSRCALPHA := 6 @@ -31,6 +33,11 @@ var _atlas: Array = [] var _terrain_attempted := false var _last_terrain_focus := Vector2(INF, INF) var _meshes: Array[MeshInstance3D] = [] +var _geometry_cache := {} +var _geometry_frame := 0 +var _bg_mesh: MeshInstance3D +var _bg_mat: StandardMaterial3D +var _bg_quad: QuadMesh # Materials keyed by texture and state, so an unchanged draw keeps its material from frame to frame. var _materials := {} @@ -39,6 +46,7 @@ var draw_count := 0 var unlit_stand_in_count := 0 # Compatibility report field; real lighting keeps this at zero. func _ready() -> void: + get_viewport().transparent_bg = true camera = Camera3D.new() camera.name = "GameCamera" camera.keep_aspect = Camera3D.KEEP_HEIGHT @@ -53,8 +61,7 @@ func _ready() -> void: add_child(light) environment = WorldEnvironment.new() environment.environment = Environment.new() - environment.environment.background_mode = Environment.BG_COLOR - environment.environment.background_color = Color.BLACK + environment.environment.background_mode = Environment.BG_CLEAR_COLOR environment.environment.ambient_light_source = Environment.AMBIENT_SOURCE_COLOR environment.environment.ambient_light_color = Color.BLACK environment.environment.tonemap_mode = Environment.TONE_MAPPER_LINEAR @@ -62,7 +69,13 @@ func _ready() -> void: func _process(_delta: float) -> void: if Metin2PythonHost.is_running(): + var profile_start := Time.get_ticks_usec() if _profile_frame else 0 update_frame() + if profile_start != 0: + profile_update_ms = float(Time.get_ticks_usec() - profile_start) / 1000.0 + +var _profile_frame := OS.has_environment("MT_PROFILE_FRAME") +var profile_update_ms := 0.0 # D3D row-vector matrix (v' = v * M) as a Godot Transform3D (v' = T * v): the rows of M are the basis # columns, row 3 is the origin. @@ -70,18 +83,8 @@ static func d3d_transform(m: PackedFloat32Array) -> Transform3D: return Transform3D(Basis(Vector3(m[0], m[1], m[2]), Vector3(m[4], m[5], m[6]), Vector3(m[8], m[9], m[10])), Vector3(m[12], m[13], m[14])) -static func multiply(a: PackedFloat32Array, b: PackedFloat32Array) -> PackedFloat32Array: - var out := PackedFloat32Array() - out.resize(16) - for r in 4: - for c in 4: - var sum := 0.0 - for k in 4: - sum += a[r * 4 + k] * b[k * 4 + c] - out[r * 4 + c] = sum - return out - func update_frame() -> void: + _geometry_frame += 1 var draws: Array = Metin2PythonHost.render3d_draws() var native_map: String = Metin2PythonHost.current_map_name() if terrain != null and not native_map.is_empty() and native_map != map_path: @@ -93,6 +96,7 @@ func update_frame() -> void: var shown := 0 var camera_set := false var light_set := false + var terrain_view := PackedFloat32Array() for draw in draws: # XYZRHW (screen-space) draws belong to the 2D pass; lines are debug geometry. if draw["pretransformed"] or draw["lines"] or draw["indices"].is_empty(): @@ -102,21 +106,25 @@ func update_frame() -> void: if proj[11] != -1.0: continue if not camera_set: - _apply_projection(proj) + _apply_projection(proj, draw) camera_set = true var view: PackedFloat32Array = draw["view"] if terrain == null and not _terrain_attempted: _load_terrain(draw) if terrain != null: - _place_terrain(view) + terrain_view = view if not light_set and draw["light0"]: _apply_light(draw, view) light_set = true var instance := _mesh_instance(shown) - instance.transform = d3d_transform(multiply(draw["world"], view)) - instance.mesh = _build_mesh(draw) + instance.transform = d3d_transform(view) * d3d_transform(draw["world"]) + instance.mesh = _mesh_for_draw(draw) + instance.material_override = _material(draw) + instance.sorting_offset = float(shown) * 0.01 instance.visible = true shown += 1 + if terrain != null and not terrain_view.is_empty(): + _place_terrain(terrain_view) for i in range(shown, _meshes.size()): _meshes[i].visible = false _meshes[i].mesh = null @@ -124,17 +132,75 @@ func update_frame() -> void: light.visible = false draw_count = shown unlit_stand_in_count = 0 + _update_background() + if _geometry_frame % 120 == 0: + for key in _geometry_cache.keys(): + if _geometry_frame - int(_geometry_cache[key][2]) > 120: + _geometry_cache.erase(key) + +func _update_background() -> void: + var bg_cmd: Dictionary = {} + for cmd in Metin2PythonHost.ui_render_commands_batched(): + if cmd.get("behind_3d", false) and cmd.get("kind", "") == "image": + bg_cmd = cmd + break + if bg_cmd.is_empty(): + if _bg_mesh != null: + _bg_mesh.visible = false + return + + if _bg_mesh == null: + _bg_mesh = MeshInstance3D.new() + _bg_mesh.name = "BackgroundQuad" + _bg_mesh.cast_shadow = GeometryInstance3D.SHADOW_CASTING_SETTING_OFF + _bg_quad = QuadMesh.new() + _bg_mesh.mesh = _bg_quad + _bg_mat = StandardMaterial3D.new() + _bg_mat.shading_mode = BaseMaterial3D.SHADING_MODE_UNSHADED + _bg_mat.depth_draw_mode = BaseMaterial3D.DEPTH_DRAW_ALWAYS + _bg_mat.cull_mode = BaseMaterial3D.CULL_DISABLED + _bg_mesh.material_override = _bg_mat + add_child(_bg_mesh) + + var bg_tex_name: String = bg_cmd.get("text", "") + var tex := UiAssets.load_tex(AssetRoot.path(), bg_tex_name) + if tex is AtlasTexture: + var at: AtlasTexture = tex + _bg_mat.albedo_texture = at.atlas + var asize: Vector2 = at.atlas.get_size() + var r: Rect2 = at.region + _bg_mat.uv1_scale = Vector3(r.size.x / asize.x, r.size.y / asize.y, 1.0) + _bg_mat.uv1_offset = Vector3(r.position.x / asize.x, r.position.y / asize.y, 0.0) + else: + _bg_mat.albedo_texture = tex + if bg_cmd.has("uv"): + var uvs: PackedVector2Array = bg_cmd["uv"] + _bg_mat.uv1_scale = Vector3(uvs[3].x - uvs[0].x, uvs[3].y - uvs[0].y, 1.0) + _bg_mat.uv1_offset = Vector3(uvs[0].x, uvs[0].y, 0.0) + else: + _bg_mat.uv1_scale = Vector3(1.0, 1.0, 1.0) + _bg_mat.uv1_offset = Vector3(0.0, 0.0, 0.0) + + var vp_size := get_viewport().get_visible_rect().size + var dist := 2800.0 + var v_size := 2.0 * dist * tan(deg_to_rad(camera.fov) / 2.0) + var h_size := v_size * (float(vp_size.x) / float(vp_size.y)) + _bg_quad.size = Vector2(h_size, v_size) + _bg_mesh.transform = Transform3D(Basis(), Vector3(0.0, 0.0, -dist)) + _bg_mesh.visible = true func _load_terrain(draw: Dictionary) -> void: if not ClassDB.class_exists("Metin2World"): return + var native_map: String = Metin2PythonHost.current_map_name() + if native_map.is_empty(): + return var camera_world := d3d_transform(draw["view"]).affine_inverse().origin var focus := Vector2(camera_world.x, -camera_world.y) if focus.distance_to(_last_terrain_focus) < 1000.0: return _last_terrain_focus = focus - var native_map: String = Metin2PythonHost.current_map_name() - map_path = native_map if not native_map.is_empty() else _map_for_position(focus) + map_path = native_map terrain = ClassDB.instantiate("Metin2World") terrain.name = "PackTerrain" terrain.set("auto_load", false) @@ -212,11 +278,38 @@ func _place_terrain(view: PackedFloat32Array) -> void: Vector3(0, -100, 0)), Vector3.ZERO) terrain.transform = d3d_transform(view) * from_map -func _apply_projection(proj: PackedFloat32Array) -> void: +func _apply_projection(proj: PackedFloat32Array, draw: Dictionary = {}) -> void: # D3DXMatrixPerspectiveFovRH: _22 = cot(fovy / 2), _33 = zf / (zn - zf), _43 = zn * zf / (zn - zf). - camera.fov = rad_to_deg(2.0 * atan(1.0 / proj[5])) - camera.near = proj[14] / proj[10] - camera.far = proj[14] / (proj[10] + 1.0) + var fov_deg := rad_to_deg(2.0 * atan(1.0 / proj[5])) + var zn: float = proj[14] / proj[10] + var zf: float = proj[14] / (proj[10] + 1.0) + camera.fov = fov_deg + camera.near = zn + camera.far = zf + + var vp_offset := Vector2.ZERO + if draw.has("viewport"): + var vp: PackedFloat32Array = draw["viewport"] + var vp_x: float = vp[0] + var vp_y: float = vp[1] + var vp_w: float = vp[2] + var vp_h: float = vp[3] + var vp_size := get_viewport().get_visible_rect().size + if vp_w > 0.0 and vp_h > 0.0 and (vp_w < vp_size.x or vp_h < vp_size.y or vp_x > 0.0 or vp_y > 0.0): + var offset_px_x: float = (vp_x + vp_w / 2.0) - (vp_size.x / 2.0) + var offset_px_y: float = (vp_y + vp_h / 2.0) - (vp_size.y / 2.0) + if absf(offset_px_x) > 0.5 or absf(offset_px_y) > 0.5: + var v_size_near := 2.0 * zn * tan(deg_to_rad(fov_deg) / 2.0) + var h_size_near := v_size_near * (vp_size.x / vp_size.y) + var off_near_x := (offset_px_x / vp_size.x) * h_size_near + var off_near_y := -(offset_px_y / vp_size.y) * v_size_near + vp_offset = Vector2(-off_near_x, off_near_y) + + var v_size_near := 2.0 * zn * tan(deg_to_rad(fov_deg) / 2.0) + if vp_offset != Vector2.ZERO: + camera.set_frustum(v_size_near, vp_offset, zn, zf) + else: + camera.set_perspective(fov_deg, zn, zf) func _apply_light(draw: Dictionary, view: PackedFloat32Array) -> void: var view_basis := d3d_transform(view).basis @@ -253,6 +346,19 @@ func _mesh_instance(index: int) -> MeshInstance3D: _meshes.append(instance) return _meshes[index] +func _mesh_for_draw(draw: Dictionary) -> ArrayMesh: + var key: int = draw.get("geometry_key", 0) + if key == 0: + return _build_mesh(draw) + var revision: int = draw.get("geometry_revision", 0) + var cached: Array = _geometry_cache.get(key, []) + if not cached.is_empty() and int(cached[0]) == revision: + cached[2] = _geometry_frame + return cached[1] + var mesh := _build_mesh(draw) + _geometry_cache[key] = [revision, mesh, _geometry_frame] + return mesh + func _build_mesh(draw: Dictionary) -> ArrayMesh: var arrays := [] arrays.resize(Mesh.ARRAY_MAX) @@ -266,15 +372,27 @@ func _build_mesh(draw: Dictionary) -> ArrayMesh: arrays[Mesh.ARRAY_INDEX] = draw["indices"] var mesh := ArrayMesh.new() mesh.add_surface_from_arrays(Mesh.PRIMITIVE_TRIANGLES, arrays) - mesh.surface_set_material(0, _material(draw)) return mesh func _material(draw: Dictionary) -> StandardMaterial3D: var lit: bool = draw["lighting"] + var z_write: int = draw.get("z_write", 1) + var tf: int = int(draw.get("texture_factor", 0xFFFFFFFF)) + var tf_a: float = float((tf >> 24) & 0xFF) / 255.0 + var tf_r: float = float((tf >> 16) & 0xFF) / 255.0 + var tf_g: float = float((tf >> 8) & 0xFF) / 255.0 + var tf_b: float = float(tf & 0xFF) / 255.0 + var factor_color := Color(tf_r, tf_g, tf_b, tf_a) + + var uses_tf: bool = bool(draw.get("uses_tf", false)) + var has_tf: bool = uses_tf and (tf != 0xFFFFFFFF and tf != -1) + var tf_key := "%02x%02x%02x%02x" % [int(tf_r * 63.0), int(tf_g * 63.0), int(tf_b * 63.0), int(tf_a * 63.0)] if has_tf else "" + var has_diffuse: bool = draw.has("diffuse") + var is_alpha: bool = draw["alpha_blend"] or (has_tf and factor_color.a < 0.99) # alpha_blend / alpha_test / lit are bools: %d needs them as ints. - var key := "%s|%d|%d|%d|%d|%d|%d|%d|%s|%s" % [draw["texture0"], int(draw["alpha_blend"]), draw["src_blend"], + var key := "%s|%d|%d|%d|%d|%d|%d|%d|%d|%d|%s|%s|%s" % [draw["texture0"], int(draw["alpha_blend"]), draw["src_blend"], draw["dest_blend"], int(draw["alpha_test"]), draw["alpha_ref"], draw["cull_mode"], int(lit), - draw["material_diffuse"], draw["material_emissive"]] + int(has_diffuse), z_write, draw["material_diffuse"], draw["material_emissive"], tf_key] if _materials.has(key): return _materials[key] var material := StandardMaterial3D.new() @@ -282,7 +400,14 @@ func _material(draw: Dictionary) -> StandardMaterial3D: if not texture_name.is_empty(): # The ported client reads its models from 40250's packs; the texture must come from there too. material.albedo_texture = UiAssets.load_pack_tex(texture_name) - material.albedo_color = draw["material_diffuse"] if lit else Color.WHITE + + var base_diffuse: Color = draw["material_diffuse"] if lit else Color.WHITE + if has_tf: + material.albedo_color = Color(base_diffuse.r * factor_color.r, base_diffuse.g * factor_color.g, + base_diffuse.b * factor_color.b, base_diffuse.a * factor_color.a) + else: + material.albedo_color = base_diffuse + material.emission_enabled = lit and draw["material_emissive"] != Color(0, 0, 0, 0) if material.emission_enabled: material.emission = draw["material_emissive"] @@ -296,18 +421,28 @@ func _material(draw: Dictionary) -> StandardMaterial3D: else BaseMaterial3D.SHADING_MODE_UNSHADED material.specular_mode = BaseMaterial3D.SPECULAR_DISABLED material.roughness = 1.0 - material.vertex_color_use_as_albedo = draw.has("diffuse") + material.vertex_color_use_as_albedo = has_diffuse match int(draw["cull_mode"]): D3DCULL_NONE: material.cull_mode = BaseMaterial3D.CULL_DISABLED D3DCULL_CW: + material.cull_mode = BaseMaterial3D.CULL_BACK + D3DCULL_CCW: material.cull_mode = BaseMaterial3D.CULL_FRONT _: material.cull_mode = BaseMaterial3D.CULL_BACK - if draw["alpha_blend"]: + if z_write == 0: + material.depth_draw_mode = BaseMaterial3D.DEPTH_DRAW_DISABLED + elif is_alpha: + material.depth_draw_mode = BaseMaterial3D.DEPTH_DRAW_ALWAYS + else: + material.depth_draw_mode = BaseMaterial3D.DEPTH_DRAW_OPAQUE_ONLY + if is_alpha: material.transparency = BaseMaterial3D.TRANSPARENCY_ALPHA if int(draw["dest_blend"]) == D3DBLEND_ONE: material.blend_mode = BaseMaterial3D.BLEND_MODE_ADD + else: + material.blend_mode = BaseMaterial3D.BLEND_MODE_MIX elif draw["alpha_test"]: material.transparency = BaseMaterial3D.TRANSPARENCY_ALPHA_SCISSOR material.alpha_scissor_threshold = float(draw["alpha_ref"]) / 255.0 diff --git a/project/python_game_render_test.gd b/project/python_game_render_test.gd index f2d6d17f..7dffcdb2 100644 --- a/project/python_game_render_test.gd +++ b/project/python_game_render_test.gd @@ -2,6 +2,7 @@ # 登录 → 选人 → 进游戏,把 RenderGame 的绘制命令(python_3d_surface.gd)画出来并截图。 # # godot --path project --script res://python_game_render_test.gd +# MT_FAKE_MOB_COUNT=24 MT_PROFILE_FRAME=1 MT_PROFILE_COMBAT=1 script/python_game_render_test.sh # # 服务器与账号只从环境变量读(MT_LIVE_HOST / MT_LIVE_LOGIN / MT_LIVE_PASSWORD,可选 MT_LIVE_AUTH_PORT # 11000、MT_LIVE_GAME_PORT 13000、MT_LIVE_SLOT 0);Python 侧也用 os.environ 取,不进源码文本。 @@ -72,8 +73,12 @@ func run() -> void: var output := OS.get_environment("MT_RENDER_OUTPUT") if output.is_empty(): output = ProjectSettings.globalize_path("res://../build/rendering/python-game-%d" % Time.get_unix_time_from_system()) + elif not output.is_absolute_path(): + output = ProjectSettings.globalize_path("res://../" + output) DirAccess.make_dir_recursive_absolute(output) var report := {"output": output, "headless": DisplayServer.get_name() == "headless"} + if OS.has_environment("MT_FAKE_MOB_COUNT"): + report["fake_mob_count"] = int(OS.get_environment("MT_FAKE_MOB_COUNT")) var host := OS.get_environment("MT_LIVE_HOST") if host.is_empty() or OS.get_environment("MT_LIVE_LOGIN").is_empty() or OS.get_environment("MT_LIVE_PASSWORD").is_empty(): @@ -132,9 +137,63 @@ func run() -> void: # The main character's draws reach the 3D surface. var drawn := in_game and await _pump_until(10, func(): return world.draw_count > 0) _check(drawn, "RenderGame draws shown (%d)" % world.draw_count) - # Let the WAIT motion run a little before the picture. - for i in 30: + # Let the WAIT motion run a little before the picture, and let GamePhase (3 packets/frame) + # finish draining the entergame burst + GC_PING when MT_FAKE_MOB_COUNT is large. + var warmup_frames := maxi(30, int(report.get("fake_mob_count", 1)) + 20) + for i in warmup_frames: await process_frame + if OS.has_environment("MT_PROFILE_FRAME"): + var frame_ms: Array[float] = [] + var ui_update_ms: Array[float] = [] + var ui_draw_ms: Array[float] = [] + var world_update_ms: Array[float] = [] + var previous := Time.get_ticks_usec() + for i in 240: + await process_frame + var current := Time.get_ticks_usec() + frame_ms.append(float(current - previous) / 1000.0) + ui_update_ms.append(ui.profile_ui_update_ms) + ui_draw_ms.append(ui.profile_draw_ms) + world_update_ms.append(world.profile_update_ms) + previous = current + frame_ms.sort() + ui_update_ms.sort() + ui_draw_ms.sort() + world_update_ms.sort() + var total := 0.0 + for sample in frame_ms: + total += sample + report["frame_profile"] = { + "samples": frame_ms.size(), + "mean_ms": total / frame_ms.size(), + "p50_ms": frame_ms[119], + "p95_ms": frame_ms[227], + "ui_update_p50_ms": ui_update_ms[119], + "ui_draw_p50_ms": ui_draw_ms[119], + "world_update_p50_ms": world_update_ms[119], + } + var command_kinds := {} + var image_texts := {} + for command in Metin2PythonHost.ui_render_commands(): + var kind: String = str(command.get("kind", "unknown")) + command_kinds[kind] = int(command_kinds.get(kind, 0)) + 1 + if kind == "image": + var image_name: String = str(command.get("text", "")) + image_texts[image_name] = int(image_texts.get(image_name, 0)) + 1 + report["frame_profile"]["ui_commands"] = command_kinds + report["frame_profile"]["image_texts"] = image_texts + var batched_commands := 0 + var glyph_batches := 0 + var batched_glyphs := 0 + for command in Metin2PythonHost.ui_render_commands_batched(): + batched_commands += 1 + if command.get("kind", "") == "glyph_batch": + glyph_batches += 1 + batched_glyphs += int(command.get("glyph_count", 0)) + report["frame_profile"]["batched_commands"] = batched_commands + report["frame_profile"]["glyph_batches"] = glyph_batches + report["frame_profile"]["batched_glyphs"] = batched_glyphs + print("FRAME_PROFILE ", report["frame_profile"]) report["draw_count"] = world.draw_count report["unlit_stand_in_count"] = world.unlit_stand_in_count report["terrain"] = world.terrain_report @@ -170,14 +229,25 @@ func run() -> void: var draws: Array = Metin2PythonHost.render3d_draws() var native_lit_draws := 0 var view_finite := true - var actor_foot_cm := INF - var foot_xy := Vector2.ZERO + # Identify character, monster and NPC draws (excluding map objects, buildings, trees and roofs). + var actor_draws: Array = [] + for draw in draws: + var tex: String = String(draw.get("texture0", "")).to_lower().replace("\\", "/") + for k in ["pc/", "pc2/", "monster/", "npc/"]: + if k in tex: + actor_draws.append(draw) + break + for draw in draws: if draw["light0"]: native_lit_draws += 1 for value in draw["view"]: if not is_finite(value): view_finite = false + + var actor_foot_cm := INF + var foot_xy := Vector2.ZERO + for draw in actor_draws: var world_xform: Transform3D = world.d3d_transform(draw["world"]) for vertex in draw["positions"]: var p: Vector3 = world_xform * vertex @@ -187,14 +257,14 @@ func run() -> void: # Every actor, not only the main one, stands on the map. An actor's parts (body, head, hair, teeth) # share its world matrix; the lowest vertex over those parts meets sample_height. var actor_low := {} - for draw in draws: + for draw in actor_draws: var draw_xform: Transform3D = world.d3d_transform(draw["world"]) var key := str(draw["world"]) for vertex in draw["positions"]: var p: Vector3 = draw_xform * vertex if not actor_low.has(key) or p.z < actor_low[key].z: actor_low[key] = p - if world.terrain: + if world.terrain and not actor_low.is_empty(): var worst_actor_gap_cm := 0.0 for low in actor_low.values(): var ground := float(world.terrain.call("sample_height", low.x / 100.0, -low.y / 100.0)) * 100.0 @@ -244,7 +314,7 @@ func run() -> void: var distinct_textures := {} for instance in world.find_children("Draw*", "MeshInstance3D", false, false): if instance.visible and instance.mesh: - var material: StandardMaterial3D = instance.mesh.surface_get_material(0) + var material: StandardMaterial3D = instance.get_active_material(0) if material and material.albedo_texture: surfaces_textured += 1 distinct_textures[material.albedo_texture] = true @@ -353,7 +423,7 @@ func run() -> void: for frame in 600: await process_frame now = dog_screen.call() - if now.distance_to(last) < 0.5: + if now.x >= 0.0 and now.distance_to(last) < 0.5: break last = now return now @@ -377,15 +447,60 @@ func run() -> void: await process_frame button.call(true, at) var mid_shot := false - var dead := await _pump_until(20, func(): + var combat_condition := func(): var now: Vector2 = dog_screen.call() if now.x >= 0.0: at = now move.call(at) - if not mid_shot and Metin2PythonHost.evaluate("33 in _target_hp") == "True" and DisplayServer.get_name() != "headless": + if not mid_shot and Metin2PythonHost.evaluate("33 in _target_hp") == "True" and DisplayServer.get_name() != "headless" and not OS.has_environment("MT_PROFILE_COMBAT"): mid_shot = true root.get_texture().get_image().save_png(output.path_join("combat_mid.png")) - return Metin2PythonHost.evaluate("_target_hp[-1:] == [0]") == "True") + return Metin2PythonHost.evaluate("_target_hp[-1:] == [0]") == "True" + var dead := false + if OS.has_environment("MT_PROFILE_COMBAT"): + var combat_frames: Array[float] = [] + var combat_ui_update: Array[float] = [] + var combat_ui_draw: Array[float] = [] + var combat_world_update: Array[float] = [] + var combat_draw_counts: Array[int] = [] + var deadline := Time.get_ticks_msec() + 20000 + var previous := Time.get_ticks_usec() + while Time.get_ticks_msec() < deadline: + if combat_condition.call(): + dead = true + break + await process_frame + var current := Time.get_ticks_usec() + combat_frames.append(float(current - previous) / 1000.0) + combat_ui_update.append(ui.profile_ui_update_ms) + combat_ui_draw.append(ui.profile_draw_ms) + combat_world_update.append(world.profile_update_ms) + combat_draw_counts.append(world.draw_count) + previous = current + combat_frames.sort() + combat_ui_update.sort() + combat_ui_draw.sort() + combat_world_update.sort() + combat_draw_counts.sort() + if not combat_frames.is_empty(): + var n := combat_frames.size() + report["combat_profile"] = { + "samples": n, + "frame_p50_ms": combat_frames[n / 2], + "frame_p95_ms": combat_frames[mini(n - 1, int(n * 0.95))], + "frame_max_ms": combat_frames[n - 1], + "ui_update_p50_ms": combat_ui_update[n / 2], + "ui_update_p95_ms": combat_ui_update[mini(n - 1, int(n * 0.95))], + "ui_draw_p50_ms": combat_ui_draw[n / 2], + "ui_draw_p95_ms": combat_ui_draw[mini(n - 1, int(n * 0.95))], + "world_update_p50_ms": combat_world_update[n / 2], + "world_update_p95_ms": combat_world_update[mini(n - 1, int(n * 0.95))], + "draw_count_p50": combat_draw_counts[n / 2], + "draw_count_max": combat_draw_counts[n - 1], + } + print("COMBAT_PROFILE ", report["combat_profile"]) + else: + dead = await _pump_until(20, combat_condition) button.call(false, at) if DisplayServer.get_name() != "headless": root.get_texture().get_image().save_png(output.path_join("combat.png")) diff --git a/project/python_ui_surface.gd b/project/python_ui_surface.gd index 58051f7a..4db298e6 100644 --- a/project/python_ui_surface.gd +++ b/project/python_ui_surface.gd @@ -8,16 +8,26 @@ const UiAssets = preload("res://ui/ui_assets.gd") const KEY_TO_DIK := { KEY_ESCAPE: 0x01, KEY_1: 0x02, KEY_2: 0x03, KEY_3: 0x04, KEY_4: 0x05, KEY_5: 0x06, KEY_6: 0x07, KEY_7: 0x08, - KEY_8: 0x09, KEY_9: 0x0a, KEY_0: 0x0b, KEY_BACKSPACE: 0x0e, - KEY_TAB: 0x0f, KEY_Q: 0x10, KEY_W: 0x11, KEY_E: 0x12, - KEY_R: 0x13, KEY_T: 0x14, KEY_Y: 0x15, KEY_U: 0x16, - KEY_I: 0x17, KEY_O: 0x18, KEY_P: 0x19, KEY_ENTER: 0x1c, - KEY_A: 0x1e, KEY_S: 0x1f, KEY_D: 0x20, KEY_F: 0x21, - KEY_G: 0x22, KEY_H: 0x23, KEY_J: 0x24, KEY_K: 0x25, - KEY_L: 0x26, KEY_Z: 0x2c, KEY_X: 0x2d, KEY_C: 0x2e, + KEY_8: 0x09, KEY_9: 0x0a, KEY_0: 0x0b, KEY_MINUS: 0x0c, + KEY_EQUAL: 0x0d, KEY_BACKSPACE: 0x0e, KEY_TAB: 0x0f, + KEY_Q: 0x10, KEY_W: 0x11, KEY_E: 0x12, KEY_R: 0x13, + KEY_T: 0x14, KEY_Y: 0x15, KEY_U: 0x16, KEY_I: 0x17, + KEY_O: 0x18, KEY_P: 0x19, KEY_BRACKETLEFT: 0x1a, KEY_BRACKETRIGHT: 0x1b, + KEY_ENTER: 0x1c, KEY_CTRL: 0x1d, KEY_A: 0x1e, KEY_S: 0x1f, + KEY_D: 0x20, KEY_F: 0x21, KEY_G: 0x22, KEY_H: 0x23, + KEY_J: 0x24, KEY_K: 0x25, KEY_L: 0x26, KEY_SEMICOLON: 0x27, + KEY_APOSTROPHE: 0x28, KEY_QUOTELEFT: 0x29, KEY_SHIFT: 0x2a, + KEY_BACKSLASH: 0x2b, KEY_Z: 0x2c, KEY_X: 0x2d, KEY_C: 0x2e, KEY_V: 0x2f, KEY_B: 0x30, KEY_N: 0x31, KEY_M: 0x32, - KEY_SPACE: 0x39, KEY_UP: 0xc8, KEY_LEFT: 0xcb, - KEY_RIGHT: 0xcd, KEY_DOWN: 0xd0, + KEY_COMMA: 0x33, KEY_PERIOD: 0x34, KEY_SLASH: 0x35, + KEY_ALT: 0x38, KEY_SPACE: 0x39, + KEY_F1: 0x3b, KEY_F2: 0x3c, KEY_F3: 0x3d, KEY_F4: 0x3e, + KEY_F5: 0x3f, KEY_F6: 0x40, KEY_F7: 0x41, KEY_F8: 0x42, + KEY_F9: 0x43, KEY_F10: 0x44, KEY_F11: 0x57, KEY_F12: 0x58, + KEY_HOME: 0xc7, KEY_UP: 0xc8, KEY_PAGEUP: 0xc9, + KEY_LEFT: 0xcb, KEY_RIGHT: 0xcd, KEY_END: 0xcf, + KEY_DOWN: 0xd0, KEY_PAGEDOWN: 0xd1, KEY_INSERT: 0xd2, + KEY_DELETE: 0xd3, } # Win32 virtual-key codes of the keys EditLine.OnIMEKeyDown handles (WM_KEYDOWN → OnIMEKeyDown); @@ -36,6 +46,11 @@ const KEY_TO_CHAR := { # "mem:" → [revision, ImageTexture]: the CGraphicFontTexture glyph pages, refetched when the # "@" of a command's name moves on (a glyph was added to the page). var _memory_textures := {} +var _tex_info_cache := {} +var _color_cache := {} +var _white_color_array := PackedColorArray([Color.WHITE]) +var _asset_root_path := "" +var _segments_used := 0 # True when this surface started system.py itself (run_app) and so owns the interpreter's lifetime. var _owns_app := false @@ -80,6 +95,8 @@ func _exit_tree() -> void: func _ready() -> void: set_anchors_and_offsets_preset(Control.PRESET_FULL_RECT) mouse_filter = Control.MOUSE_FILTER_STOP + texture_repeat = CanvasItem.TEXTURE_REPEAT_ENABLED + _asset_root_path = AssetRoot.path() _sync_size() func _notification(what: int) -> void: @@ -100,7 +117,10 @@ func _process(_delta: float) -> void: get_tree().quit(0) return if Metin2PythonHost.is_running(): + var profile_start := Time.get_ticks_usec() if _profile_frame else 0 Metin2PythonHost.ui_update() + if profile_start != 0: + profile_ui_update_ms = float(Time.get_ticks_usec() - profile_start) / 1000.0 queue_redraw() func _gui_input(event: InputEvent) -> void: @@ -112,10 +132,30 @@ func _gui_input(event: InputEvent) -> void: if event.button_index >= MOUSE_BUTTON_LEFT and event.button_index <= MOUSE_BUTTON_MIDDLE: Metin2PythonHost.ui_mouse_button(event.button_index, event.pressed, int(event.position.x), int(event.position.y)) + elif event.pressed: + if event.button_index == MOUSE_BUTTON_WHEEL_UP: + var factor: float = event.factor if event.factor > 0.0 else 1.0 + Metin2PythonHost.ui_mouse_wheel(int(round(120.0 * factor))) + elif event.button_index == MOUSE_BUTTON_WHEEL_DOWN: + var factor: float = event.factor if event.factor > 0.0 else 1.0 + Metin2PythonHost.ui_mouse_wheel(-int(round(120.0 * factor))) + elif event is InputEventPanGesture: + # macOS trackpad two-finger scroll: scrolling up has negative delta.y, zooming in + var delta: int = -int(round(event.delta.y * 30.0)) + if delta != 0: + Metin2PythonHost.ui_mouse_wheel(delta) + elif event is InputEventMagnifyGesture: + # macOS trackpad pinch to zoom: factor > 1.0 is zoom in + var delta: int = int(round((event.factor - 1.0) * 600.0)) + if delta != 0: + Metin2PythonHost.ui_mouse_wheel(delta) func _unhandled_key_input(event: InputEvent) -> void: if event is InputEventKey and Metin2PythonHost.is_running(): - var dik: int = KEY_TO_DIK.get(event.physical_keycode, 0) + var code: int = event.physical_keycode + var dik: int = KEY_TO_DIK.get(code, 0) + if dik == 0 and event.keycode != 0: + dik = KEY_TO_DIK.get(event.keycode, 0) if dik != 0: Metin2PythonHost.ui_key(dik, event.pressed) if event.pressed: @@ -159,79 +199,247 @@ var _segments: Array[RID] = [] var _materials: Array[RID] = [] var _mask_shader := RID() var _white: ImageTexture +var _profile_frame := OS.has_environment("MT_PROFILE_FRAME") +var profile_ui_update_ms := 0.0 +var profile_draw_ms := 0.0 + +func _get_color_array(argb: int) -> PackedColorArray: + if argb == 0xFFFFFFFF or argb == -1: + return _white_color_array + var arr: PackedColorArray = _color_cache.get(argb, PackedColorArray()) + if not arr.is_empty(): + return arr + var c := Color8((argb >> 16) & 255, (argb >> 8) & 255, argb & 255, (argb >> 24) & 255) + arr = PackedColorArray([c]) + _color_cache[argb] = arr + return arr + +func _get_tex_draw_info(name: String) -> Array: + var info: Array = _tex_info_cache.get(name, []) + if not info.is_empty(): + return info + var texture: Texture2D = _texture(name) + if texture == null: + _tex_info_cache[name] = [] + return [] + var tex_rid: RID = texture.get_rid() + var is_atlas := false + var u0 := 0.0 + var v0 := 0.0 + var du := 1.0 + var dv := 1.0 + var default_uv := PackedVector2Array() + if texture is AtlasTexture: + var at: AtlasTexture = texture + if at.atlas != null: + tex_rid = at.atlas.get_rid() + var asize: Vector2 = at.atlas.get_size() + if asize.x > 0 and asize.y > 0: + is_atlas = true + var r: Rect2 = at.region + u0 = r.position.x / asize.x + v0 = r.position.y / asize.y + du = r.size.x / asize.x + dv = r.size.y / asize.y + default_uv = PackedVector2Array([ + Vector2(u0, v0), + Vector2(u0 + du, v0), + Vector2(u0 + du, v0 + dv), + Vector2(u0, v0 + dv) + ]) + if default_uv.is_empty(): + default_uv = PackedVector2Array([ + Vector2(0, 0), + Vector2(1, 0), + Vector2(1, 1), + Vector2(0, 1) + ]) + info = [tex_rid, is_atlas, u0, v0, du, dv, default_uv, texture] + _tex_info_cache[name] = info + return info func _draw() -> void: - var used := 0 + var profile_start := Time.get_ticks_usec() if _profile_frame else 0 + var fg_used := 0 + var fg_plain := false var masked := 0 - var plain := false - for segment in _segments: - RenderingServer.canvas_item_clear(segment) + + for i in _segments_used: + RenderingServer.canvas_item_clear(_segments[i]) + _segments_used = 0 + if not Metin2PythonHost.is_running(): return - for command in Metin2PythonHost.ui_render_commands(): - var argb: int = command["argb"] - var color := Color8((argb >> 16) & 255, (argb >> 8) & 255, - argb & 255, (argb >> 24) & 255) - var p1 := Vector2(command["x1"], command["y1"]) - var p2 := Vector2(command["x2"], command["y2"]) - var clip := Rect2(Vector2(command["clip_x1"], command["clip_y1"]), - Vector2(command["clip_x2"] - command["clip_x1"], - command["clip_y2"] - command["clip_y1"])) - if command["kind"] == "bar": - var visible := Rect2(p1, p2 - p1).intersection(clip) - if visible.has_area(): - if not plain: - used = _segment(used, RID()) - plain = true - RenderingServer.canvas_item_add_rect(_segments[used - 1], visible, color) - elif command["kind"] == "gradient_bar": - var visible := Rect2(p1, p2 - p1).intersection(clip) - if visible.has_area(): - if not plain: - used = _segment(used, RID()) - plain = true - var bottom_argb: int = command["end_argb"] + + var has_3d: bool = Metin2PythonHost.has_3d_draws() + + for command in Metin2PythonHost.ui_render_commands_batched(): + # Skip background UI commands when 3D is active, as the 3D surface renders the background Quad + if has_3d and bool(command.get(&"behind_3d", false)): + continue + var kind: Variant = command[&"kind"] + if kind == &"glyph_batch": + var info := _get_tex_draw_info(command[&"text"]) + if info.is_empty(): + continue + if not fg_plain: + fg_used = _segment(fg_used, RID()) + fg_plain = true + RenderingServer.canvas_item_add_triangle_array(_segments[fg_used - 1], + command[&"indices"], command[&"points"], command[&"colors"], + command[&"uvs"], PackedInt32Array(), PackedFloat32Array(), info[0]) + continue + + var x1: float = command[&"x1"] + var y1: float = command[&"y1"] + var x2: float = command[&"x2"] + var y2: float = command[&"y2"] + var cx1: float = command[&"clip_x1"] + var cy1: float = command[&"clip_y1"] + var cx2: float = command[&"clip_x2"] + var cy2: float = command[&"clip_y2"] + + # 1) Completely culled by scissor? + if x2 <= cx1 or x1 >= cx2 or y2 <= cy1 or y1 >= cy2: + continue + + var argb: int = command[&"argb"] + var target_segments := _segments + + if kind == &"image": + var text_name: String = command.get(&"text", "") + var mask_name: String = command.get(&"mask", "") + var mask_tex: Texture2D = _texture(mask_name) if not mask_name.is_empty() else null + if text_name.is_empty() and mask_tex != null: + text_name = "__white__" + + var info := _get_tex_draw_info(text_name) + if info.is_empty(): + continue + var tex_rid: RID = info[0] + var is_atlas: bool = info[1] + var u0: float = info[2] + var v0: float = info[3] + var du: float = info[4] + var dv: float = info[5] + + var quad: PackedVector2Array = command[&"quad"] + var quad_uv: PackedVector2Array = command[&"uv"] + + var needs_clip := x1 < cx1 or x2 > cx2 or y1 < cy1 or y2 > cy2 + var poly_pts: PackedVector2Array + var poly_uv: PackedVector2Array + + if not needs_clip: + poly_pts = PackedVector2Array([quad[0], quad[1], quad[3], quad[2]]) + var is_full_uv := quad_uv[0] == Vector2.ZERO and quad_uv[1] == Vector2(1, 0) and quad_uv[2] == Vector2(0, 1) and quad_uv[3] == Vector2.ONE + if is_full_uv: + poly_uv = info[6] + elif is_atlas: + poly_uv = PackedVector2Array([ + Vector2(u0 + quad_uv[0].x * du, v0 + quad_uv[0].y * dv), + Vector2(u0 + quad_uv[1].x * du, v0 + quad_uv[1].y * dv), + Vector2(u0 + quad_uv[3].x * du, v0 + quad_uv[3].y * dv), + Vector2(u0 + quad_uv[2].x * du, v0 + quad_uv[2].y * dv) + ]) + else: + poly_uv = PackedVector2Array([quad_uv[0], quad_uv[1], quad_uv[3], quad_uv[2]]) + else: + var mapped_uv: PackedVector2Array + if is_atlas: + mapped_uv = PackedVector2Array([ + Vector2(u0 + quad_uv[0].x * du, v0 + quad_uv[0].y * dv), + Vector2(u0 + quad_uv[1].x * du, v0 + quad_uv[1].y * dv), + Vector2(u0 + quad_uv[2].x * du, v0 + quad_uv[2].y * dv), + Vector2(u0 + quad_uv[3].x * du, v0 + quad_uv[3].y * dv) + ]) + else: + mapped_uv = quad_uv + var clip_rect := Rect2(cx1, cy1, cx2 - cx1, cy2 - cy1) + var clipped := _clip_quad(quad, mapped_uv, clip_rect) + if clipped[0].size() < 3: + continue + poly_pts = clipped[0] + poly_uv = clipped[1] + + if mask_tex != null: + var mat_rid := _mask_material(masked, mask_tex, quad_uv, command[&"mask_uv"]) + fg_used = _segment(fg_used, mat_rid) + fg_plain = false + masked += 1 + elif not fg_plain: + fg_used = _segment(fg_used, RID()) + fg_plain = true + var seg_idx := fg_used - 1 + RenderingServer.canvas_item_add_polygon(target_segments[seg_idx], poly_pts, + _get_color_array(argb), poly_uv, tex_rid) + + elif kind == &"bar": + var bx1 := maxf(x1, cx1) + var by1 := maxf(y1, cy1) + var bx2 := minf(x2, cx2) + var by2 := minf(y2, cy2) + if bx2 > bx1 and by2 > by1: + if not fg_plain: + fg_used = _segment(fg_used, RID()) + fg_plain = true + var seg_idx := fg_used - 1 + var color := Color8((argb >> 16) & 255, (argb >> 8) & 255, + argb & 255, (argb >> 24) & 255) + RenderingServer.canvas_item_add_rect(target_segments[seg_idx], + Rect2(bx1, by1, bx2 - bx1, by2 - by1), color) + + elif kind == &"gradient_bar": + var bx1 := maxf(x1, cx1) + var by1 := maxf(y1, cy1) + var bx2 := minf(x2, cx2) + var by2 := minf(y2, cy2) + if bx2 > bx1 and by2 > by1: + if not fg_plain: + fg_used = _segment(fg_used, RID()) + fg_plain = true + var seg_idx := fg_used - 1 + var color := Color8((argb >> 16) & 255, (argb >> 8) & 255, + argb & 255, (argb >> 24) & 255) + var bottom_argb: int = command[&"end_argb"] var bottom := Color8((bottom_argb >> 16) & 255, (bottom_argb >> 8) & 255, bottom_argb & 255, (bottom_argb >> 24) & 255) - var top_factor := (visible.position.y - p1.y) / (p2.y - p1.y) - var bottom_factor := (visible.end.y - p1.y) / (p2.y - p1.y) - RenderingServer.canvas_item_add_polygon(_segments[used - 1], - PackedVector2Array([visible.position, Vector2(visible.end.x, visible.position.y), - visible.end, Vector2(visible.position.x, visible.end.y)]), + var top_factor := (by1 - y1) / (y2 - y1) if y2 != y1 else 0.0 + var bottom_factor := (by2 - y1) / (y2 - y1) if y2 != y1 else 1.0 + RenderingServer.canvas_item_add_polygon(target_segments[seg_idx], + PackedVector2Array([Vector2(bx1, by1), Vector2(bx2, by1), + Vector2(bx2, by2), Vector2(bx1, by2)]), PackedColorArray([color.lerp(bottom, top_factor), color.lerp(bottom, top_factor), color.lerp(bottom, bottom_factor), color.lerp(bottom, bottom_factor)])) - elif command["kind"] == "line": - var segment := _clip_line(p1, p2, clip) + + elif kind == &"line": + var clip_rect := Rect2(cx1, cy1, cx2 - cx1, cy2 - cy1) + var p1 := Vector2(x1, y1) + var p2 := Vector2(x2, y2) + var segment := _clip_line(p1, p2, clip_rect) if segment.size() == 2: - if not plain: - used = _segment(used, RID()) - plain = true - RenderingServer.canvas_item_add_line(_segments[used - 1], segment[0], segment[1], color) - elif command["kind"] == "image": - var texture: Texture2D = _texture(command["text"]) - var mask: Texture2D = _texture(command["mask"]) if command.has("mask") else null - if mask != null and texture == null: - texture = _white_texture() - if texture != null: - var clipped := _clip_quad(command["quad"], command["uv"], clip) - if clipped[0].size() >= 3: - if mask != null: - used = _segment(used, _mask_material(masked, mask, command["uv"], command["mask_uv"])) - masked += 1 - plain = false - elif not plain: - used = _segment(used, RID()) - plain = true - RenderingServer.canvas_item_add_polygon(_segments[used - 1], clipped[0], - PackedColorArray([color]), clipped[1], texture.get_rid()) + if not fg_plain: + fg_used = _segment(fg_used, RID()) + fg_plain = true + var seg_idx := fg_used - 1 + var color := Color8((argb >> 16) & 255, (argb >> 8) & 255, + argb & 255, (argb >> 24) & 255) + RenderingServer.canvas_item_add_line(target_segments[seg_idx], segment[0], segment[1], color) + + _segments_used = fg_used + if profile_start != 0: + profile_draw_ms = float(Time.get_ticks_usec() - profile_start) / 1000.0 # Opens the next pooled child item (draw order = index) with the given material; returns the count used. func _segment(used: int, material: RID) -> int: - if used == _segments.size(): + var pool := _segments + var parent_item: RID = get_canvas_item() + if used == pool.size(): var item := RenderingServer.canvas_item_create() - RenderingServer.canvas_item_set_parent(item, get_canvas_item()) - _segments.push_back(item) - var item := _segments[used] + RenderingServer.canvas_item_set_parent(item, parent_item) + RenderingServer.canvas_item_set_default_texture_repeat(item, RenderingServer.CANVAS_ITEM_TEXTURE_REPEAT_ENABLED) + pool.push_back(item) + var item := pool[used] RenderingServer.canvas_item_set_draw_index(item, used) RenderingServer.canvas_item_set_material(item, material) return used + 1 @@ -260,8 +468,11 @@ func _white_texture() -> Texture2D: return _white func _texture(name: String) -> Texture2D: + if name == "__white__": + return _white_texture() if not name.begins_with("mem:"): - return UiAssets.load_tex(AssetRoot.path(), name) + var root := _asset_root_path if not _asset_root_path.is_empty() else AssetRoot.path() + return UiAssets.load_tex(root, name) var at := name.find("@") var key := name.substr(0, at) if at >= 0 else name var revision := int(name.substr(at + 1)) if at >= 0 else 0 @@ -292,6 +503,9 @@ func _clip_quad(quad: PackedVector2Array, uv: PackedVector2Array, clip: Rect2) - var det := edge_u.cross(edge_v) if is_zero_approx(det): return [PackedVector2Array(), PackedVector2Array()] + if clip.has_point(quad[0]) and clip.has_point(quad[1]) \ + and clip.has_point(quad[2]) and clip.has_point(quad[3]): + return [outline, PackedVector2Array([uv[0], uv[1], uv[3], uv[2]])] var rect := PackedVector2Array([clip.position, Vector2(clip.end.x, clip.position.y), clip.end, Vector2(clip.position.x, clip.end.y)]) var pieces := Geometry2D.intersect_polygons(outline, rect) diff --git a/project/sample_profile.gd b/project/sample_profile.gd new file mode 100644 index 00000000..71311357 --- /dev/null +++ b/project/sample_profile.gd @@ -0,0 +1,14 @@ +@tool +extends SceneTree + +func _init() -> void: + var tw = Metin2World.new() + tw.set("assets_root", "pack://") + tw.set("map_path", "metin2_map_a1") + tw.set("auto_load", false) + tw.call("set_focus_tile", Vector2i(2, 2)) + if tw.call("load_map"): + for y in range(68200, 66000, -100): + var h = float(tw.call("sample_height", 59700.0 / 100.0, float(y) / 100.0)) * 100.0 + print("y = %d -> height = %.2f" % [y, h]) + quit(0) diff --git a/project/sample_profile.gd.uid b/project/sample_profile.gd.uid new file mode 100644 index 00000000..6b64993e --- /dev/null +++ b/project/sample_profile.gd.uid @@ -0,0 +1 @@ +uid://bt4n4goaei8mi diff --git a/project/test_actor_heights.gd b/project/test_actor_heights.gd new file mode 100644 index 00000000..f3c9df2b --- /dev/null +++ b/project/test_actor_heights.gd @@ -0,0 +1,107 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + world = Python3DSurface.new() + world.name = "Python3DSurface" + ui = PythonUISurface.new() + ui.name = "PythonUISurface" + root.add_child(ui) + root.add_child(world) + + var err: String = ui.run_app() + if err != "": + print("ERROR: run_app: ", err) + quit(1) + return + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect, game, player, background, chr, chrmgr\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + var host = OS.get_environment("MT_LIVE_HOST") + var login = OS.get_environment("MT_LIVE_LOGIN") + var pwd = OS.get_environment("MT_LIVE_PASSWORD") + var auth_port = int(OS.get_environment("MT_LIVE_AUTH_PORT")) + var game_port = int(OS.get_environment("MT_LIVE_GAME_PORT")) + + _py("_stream.SetConnectInfo('%s', %d, '%s', %d)\n" % [host, game_port, host, auth_port] + + "_stream.curPhaseWindow.Connect('%s', '%s')" % [login, pwd]) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + _py("_stream.curPhaseWindow.SelectSlot(0)\n_stream.curPhaseWindow.StartGame()") + + await _pump_until(60, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, game.GameWindow) and _stream.curPhaseWindow.IsShow()" + )) + + await _pump_until(10, func(): return world.draw_count > 0) + for i in 30: + await process_frame + + var draws: Array = Metin2PythonHost.render3d_draws() + print("Game 3D draws count: ", draws.size()) + + # Find actual actors + var actor_draws: Array = [] + for draw in draws: + var tex = draw.get("texture0", "").to_lower() + var is_actor = false + for k in ["pc/", "pc2/", "monster/", "npc/", "warrior", "assassin", "sura", "shaman", "dog", "wolf"]: + if k in tex: + is_actor = true + break + if is_actor: + actor_draws.append(draw) + + print("Actual actor draws count: ", actor_draws.size()) + + var actor_low := {} + for draw in actor_draws: + var draw_xform: Transform3D = world.d3d_transform(draw["world"]) + var key := str(draw["world"]) + for vertex in draw["positions"]: + var p: Vector3 = draw_xform * vertex + if not actor_low.has(key) or p.z < actor_low[key].z: + actor_low[key] = p + + print("Distinct actors count: ", actor_low.size()) + if world.terrain: + var worst_actor_gap_cm := 0.0 + for key in actor_low: + var low: Vector3 = actor_low[key] + var ground := float(world.terrain.call("sample_height", low.x / 100.0, -low.y / 100.0)) * 100.0 + var gap = absf(low.z - ground) + print("Actor at (%f, %f): foot_z=%f, ground_z=%f, gap=%f cm" % [low.x, low.y, low.z, ground, gap]) + worst_actor_gap_cm = maxf(worst_actor_gap_cm, gap) + print("WORST ACTUAL ACTOR GAP: ", worst_actor_gap_cm, " cm") + + quit(0) diff --git a/project/test_actor_heights.gd.uid b/project/test_actor_heights.gd.uid new file mode 100644 index 00000000..2207d789 --- /dev/null +++ b/project/test_actor_heights.gd.uid @@ -0,0 +1 @@ +uid://cmum40u4abuep diff --git a/project/test_basis.gd b/project/test_basis.gd new file mode 100644 index 00000000..ab1f0dc6 --- /dev/null +++ b/project/test_basis.gd @@ -0,0 +1,13 @@ +@tool +extends SceneTree + +func _init() -> void: + var b := Basis(Vector3(1, 2, 3), Vector3(4, 5, 6), Vector3(7, 8, 9)) + print("b.x = ", b.x) + print("b.y = ", b.y) + print("b.z = ", b.z) + var v := Vector3(1, 0, 0) + print("b * Vector3(1, 0, 0) = ", b * v) + var v2 := Vector3(0, 1, 0) + print("b * Vector3(0, 1, 0) = ", b * v2) + quit(0) diff --git a/project/test_basis.gd.uid b/project/test_basis.gd.uid new file mode 100644 index 00000000..dc4ac18f --- /dev/null +++ b/project/test_basis.gd.uid @@ -0,0 +1 @@ +uid://d17rcsnfyx418 diff --git a/project/test_select_character.gd b/project/test_select_character.gd new file mode 100644 index 00000000..f698b1b2 --- /dev/null +++ b/project/test_select_character.gd @@ -0,0 +1,95 @@ +extends SceneTree + +const PythonUISurface = preload("res://python_ui_surface.gd") +const Python3DSurface = preload("res://python_3d_surface.gd") + +var ui: Control +var world: Node3D + +func _initialize() -> void: + call_deferred("run") + +func _py(source: String) -> bool: + return Metin2PythonHost.run_line(source) == "" + +func _key(keycode: Key, unicode: int, pressed: bool) -> void: + var event := InputEventKey.new() + event.keycode = keycode + event.physical_keycode = keycode + event.unicode = unicode + event.pressed = pressed + root.push_input(event) + +func _press(keycode: Key) -> void: + _key(keycode, 0, true) + _key(keycode, 0, false) + +func _type(text: String) -> void: + for i in text.length(): + var ch := text.unicode_at(i) + var keycode := OS.find_keycode_from_string(text[i].to_upper()) + _key(keycode, ch, true) + _key(keycode, ch, false) + +func _pump_until(seconds: float, condition: Callable) -> bool: + var deadline := Time.get_ticks_msec() + int(seconds * 1000.0) + while Time.get_ticks_msec() < deadline: + if condition.call(): + return true + await process_frame + return condition.call() + +func run() -> void: + root.size = Vector2i(1024, 768) + + world = Python3DSurface.new() + world.name = "Python3DSurface" + ui = PythonUISurface.new() + ui.name = "PythonUISurface" + root.add_child(ui) + root.add_child(world) + + var err: String = ui.run_app() + if err != "": + quit(1) + return + + await _pump_until(15, func(): return _py( + "import __main__, os, networkModule, introLogin, introSelect\n" + + "__main__._stream = [o for o in __import__('gc').get_objects() if isinstance(o, networkModule.MainStream)][0]" + )) + + await _pump_until(20, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introLogin.LoginWindow)" + )) + + _py( + "_env = os.environ\n" + + "_stream.SetConnectInfo(_env['MT_LIVE_HOST'], int(_env.get('MT_LIVE_GAME_PORT', '13000')), " + + "_env['MT_LIVE_HOST'], int(_env.get('MT_LIVE_AUTH_PORT', '11000')))\n" + + "_w = _stream.curPhaseWindow\n" + + "_w._LoginWindow__OpenLoginBoard()\n" + + "_w.idEditLine.SetText('')\n" + + "_w.pwdEditLine.SetText('')\n" + + "_w.idEditLine.SetFocus()" + ) + _type(OS.get_environment("MT_LIVE_LOGIN")) + _press(KEY_TAB) + _type(OS.get_environment("MT_LIVE_PASSWORD")) + _py("del _env, _w") + _press(KEY_ENTER) + + await _pump_until(30, func(): return _py( + "assert isinstance(_stream.curPhaseWindow, introSelect.SelectCharacterWindow)" + )) + + for i in 15: + await process_frame + + var img_final = root.get_texture().get_image() + img_final.save_png("/tmp/select_character_final.png") + print("Saved /tmp/select_character_final.png") + + quit(0) + + diff --git a/project/test_select_character.gd.uid b/project/test_select_character.gd.uid new file mode 100644 index 00000000..8c489c20 --- /dev/null +++ b/project/test_select_character.gd.uid @@ -0,0 +1 @@ +uid://dpi4fhed4wc28 diff --git a/project/ui/ui_assets.gd b/project/ui/ui_assets.gd index e07d5b89..dc2d7ec3 100644 --- a/project/ui/ui_assets.gd +++ b/project/ui/ui_assets.gd @@ -1,10 +1,10 @@ -# UiAssets (P1) —— 解析 uiscript 里的图片路径(`.sub` / `.tga` / `.png` / `.jpg`)。 +# UiAssets (P1) —— 解析 uiscript 里的图片路径(`.sub` / `.tga` / `.png` / `.jpg` / `.dds`)。 # # 路径形如 `d:/ymir work/ui/public/middle_button_01.sub`。解析: -# 1) 去掉盘符前缀 -# 2) 依次试: / · //(散包目录) -# 3) `.sub` = 子图描述(title/image/left/top/right/bottom)→ 载入其 image + 裁剪成 AtlasTexture -# `.dds` 走 Metin2World.load_dds(C++ dxt 解码)运行时解成 Image。 +# 1) 优先从 40250 Client/pack(Metin2Pack)取(保持 d:/ 盘符或规范相对路径) +# 2) 支持 pack 内 .sub 子图切片(AtlasTexture),支持 version 1.0 / 2.0 及公用 Public.dds / IntroEmpire.dds 母图寻址 +# 3) 若 pack 找不到且指定了 assets_root,回退到散文件目录 +# 4) .dds 走 Metin2World.load_dds(C++ dxt 解码)运行时解成 Image。 extends RefCounted static var _cache := {} @@ -19,21 +19,20 @@ static func load_dds_image(path: String) -> Image: return null if not ClassDB.class_exists("Metin2World"): return null - # Metin2World is a Node, not RefCounted; a static reference never frees it. - # The decoder returns an independently owned Image, so release the helper - # after each decode (textures themselves remain cached by load_tex). var helper: Object = ClassDB.instantiate("Metin2World") var img = helper.call("load_dds", path) helper.free() return img if img is Image else null static func load_tex(assets_root: String, vpath: String) -> Texture2D: - if vpath == "" or assets_root == "": + if vpath == "": return null - var key := assets_root + "|" + vpath + var key := (assets_root if assets_root != "" else "pack") + "|" + vpath if _cache.has(key): return _cache[key] - var tex: Texture2D = _load_uncached(assets_root, vpath) + var tex: Texture2D = _load_pack_any(vpath) + if tex == null and assets_root != "": + tex = _load_uncached(assets_root, vpath) _cache[key] = tex return tex @@ -45,8 +44,7 @@ static func load_pack_tex(vpath: String) -> Texture2D: var key := "pack|" + vpath if _cache.has(key): return _cache[key] - # 40250 的包名带盘符(d:/ymir work/...),原样交给 CEterPackManager。 - var tex: Texture2D = _load_pack_image(vpath.replace("\\", "/")) + var tex: Texture2D = _load_pack_any(vpath) _cache[key] = tex return tex @@ -56,6 +54,93 @@ static func _strip_drive(p: String) -> String: s = s.substr(2) return s.lstrip("/") +static func _find_pack_path(vpath: String) -> String: + if not ClassDB.class_exists("Metin2Pack") or not Metin2Pack.is_ready(): + return "" + var norm := vpath.replace("\\", "/").strip_edges() + var stripped := _strip_drive(norm) + var candidates: Array[String] = [ + norm, + stripped, + "d:/" + stripped, + norm.to_lower(), + stripped.to_lower(), + ("d:/" + stripped).to_lower(), + ] + for c in candidates: + if Metin2Pack.exists(c): + return c + return "" + +static func _load_pack_any(vpath: String) -> Texture2D: + var pack_path := _find_pack_path(vpath) + if pack_path == "": + return null + var ext := pack_path.get_extension().to_lower() + if ext == "sub": + return _load_pack_sub(pack_path) + return _load_pack_image(pack_path) + +static func _load_pack_sub(pack_path: String) -> Texture2D: + if not ClassDB.class_exists("Metin2Pack") or not Metin2Pack.is_ready(): + return null + if not Metin2Pack.exists(pack_path): + return null + var bytes: PackedByteArray = Metin2Pack.get_bytes(pack_path) + if bytes.is_empty(): + return null + var txt := bytes.get_string_from_ascii() + var image_name := "" + var version := "" + var l := 0 + var t := 0 + var r := -1 + var b := -1 + for line in txt.split("\n"): + var parts := line.strip_edges().split(" ", false) + if parts.size() < 2: + continue + match parts[0].to_lower(): + "version": version = parts[1].strip_edges().trim_prefix('"').trim_suffix('"') + "image": image_name = parts[1].strip_edges().trim_prefix('"').trim_suffix('"') + "left": l = int(parts[1]) + "top": t = int(parts[1]) + "right": r = int(parts[1]) + "bottom": b = int(parts[1]) + if image_name == "": + return null + + var sub_dir := pack_path.get_base_dir() + var candidates: Array[String] = [] + if version == "2.0": + candidates.append(sub_dir.path_join(image_name)) + candidates.append("d:/ymir work/ui/".path_join(image_name)) + else: + candidates.append("d:/ymir work/ui/".path_join(image_name)) + candidates.append(sub_dir.path_join(image_name)) + + var search_dir := sub_dir + for _i in 4: + var parent := search_dir.get_base_dir() + if parent == search_dir or parent.is_empty(): + break + candidates.append(parent.path_join(image_name)) + search_dir = parent + + var base_tex: Texture2D = null + for cand in candidates: + base_tex = _load_pack_any(cand) + if base_tex != null: + break + if base_tex == null: + return null + if r <= l or b <= t: + return base_tex + var at := AtlasTexture.new() + at.atlas = base_tex + at.region = Rect2(l, t, r - l, b - t) + return at + static func _resolve(assets_root: String, rel: String) -> String: var direct := assets_root.path_join(rel) if FileAccess.file_exists(direct): @@ -67,20 +152,18 @@ static func _resolve(assets_root: String, rel: String) -> String: var cand := assets_root.path_join(sub).path_join(rel) if FileAccess.file_exists(cand): return cand - # 大小写不敏感兜底:/**/ymir work/ui/... —— 只按 basename 找 return "" static func _load_uncached(assets_root: String, vpath: String) -> Texture2D: var rel := _strip_drive(vpath) var real := _resolve(assets_root, rel) if real == "": - # 试把 .sub 换成 .tga / .png for ext: String in [".tga", ".png", ".jpg"]: real = _resolve(assets_root, rel.get_basename() + ext) if real != "": break if real == "": - return _load_pack_image(rel) + return _load_pack_any(vpath) if real.get_extension().to_lower() == "sub": return _load_sub(real) return _load_image_file(real) @@ -112,8 +195,6 @@ static func _load_pack_image(rel: String) -> Texture2D: return ImageTexture.create_from_image(image) if image != null and not image.is_empty() else null static func _load_image_file(path: String) -> Texture2D: - # .sub 文件可能引用了未随当前资源包发布的共享贴图(例如 Public.tga)。 - # 先做存在性检查,避免 headless/UI fallback 因缺失可选贴图刷错误日志。 if not FileAccess.file_exists(path): return null var ext := path.get_extension().to_lower() @@ -144,8 +225,6 @@ static func _load_sub(path: String) -> Texture2D: "bottom": b = int(parts[1]) if image_name == "": return null - # Skill .sub files live below ui/skill/, while their shared DDS - # lives in ui/. Search the containing directory and its parents. var img_path := "" var search_dir := path.get_base_dir() for _i in 6: diff --git a/script/native_mac_acceptance.mjs b/script/native_mac_acceptance.mjs new file mode 100644 index 00000000..789e8729 --- /dev/null +++ b/script/native_mac_acceptance.mjs @@ -0,0 +1,205 @@ +#!/usr/bin/env node +// Run the macOS native client with a real server or the local fake server. +// A real-server run remains unverified until a person checks the in-game flows. +import fs from 'node:fs'; +import path from 'node:path'; +import net from 'node:net'; +import { spawn, execFile } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; +import { redact, secretLiterals } from './redact_stream.mjs'; + +const repo = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const defaults = { + binary: path.join(repo, 'build-release/native_render/mt_native_render'), + client: path.join(repo, 'Client'), + output: path.join(repo, 'build/native-acceptance', new Date().toISOString().replaceAll(':', '-')), +}; + +function usage() { + console.log(`Usage: node script/native_mac_acceptance.mjs (--server HOST:AUTH_PORT:GAME_PORT | --fake) [options] + --binary FILE Native live-client binary (default: build-release/native_render/mt_native_render) + --client DIR 40250 Client directory (default: Client) + --output DIR New evidence directory (default: build/native-acceptance/) + --min-seconds N Minimum real-server observation time (default: 1800) + --frames N Fake-server smoke frames (default: 180; fake mode only) + --help Show this help + +Real-server mode opens the login screen. Enter credentials in the app, exercise +the checklist in the report, then close the window. Credentials are never +passed on the command line. The report requires manual review before PASS.`); +} + +function optionsOf(argv) { + const options = { ...defaults, minSeconds: 1800, frames: 180 }; + for (let i = 0; i < argv.length; i += 1) { + const key = argv[i]; + if (key === '--help') { options.help = true; continue; } + if (!['--server', '--fake', '--binary', '--client', '--output', '--min-seconds', '--frames'].includes(key)) + throw new Error(`Unknown option: ${key}`); + if (key === '--fake') { options.fake = true; continue; } + if (++i >= argv.length) throw new Error(`Missing value for ${key}`); + const field = { '--server': 'server', '--binary': 'binary', '--client': 'client', + '--output': 'output', '--min-seconds': 'minSeconds', '--frames': 'frames' }[key]; + options[field] = argv[i]; + } + if (!options.help) { + if (Boolean(options.server) === Boolean(options.fake)) throw new Error('Choose exactly one of --server or --fake'); + for (const field of ['minSeconds', 'frames']) { + options[field] = Number(options[field]); + if (!Number.isInteger(options[field]) || options[field] < (field === 'frames' ? 1 : 0)) + throw new Error(`Invalid ${field}`); + } + if (options.server && argv.includes('--frames')) throw new Error('--frames is only for fake-server smoke'); + } + return options; +} + +function parseServer(spec) { + const match = /^([^:\s]+):([0-9]+):([0-9]+)$/.exec(spec); + if (!match || [match[2], match[3]].some((value) => +value < 1 || +value > 65535)) + throw new Error('Server must be HOST:AUTH_PORT:GAME_PORT (one shared host)'); + return { host: match[1], authPort: +match[2], gamePort: +match[3] }; +} + +function reachable(host, port) { + return new Promise((resolve) => { + const socket = net.createConnection({ host, port }); + socket.setTimeout(3000); + socket.once('connect', () => { socket.destroy(); resolve(true); }); + socket.once('error', () => { socket.destroy(); resolve(false); }); + socket.once('timeout', () => { socket.destroy(); resolve(false); }); + }); +} + +function readRss(pid) { + return new Promise((resolve) => execFile('ps', ['-o', 'stat=,rss=', '-p', String(pid)], + { timeout: 3000 }, (error, stdout) => { + if (error) return resolve(null); + const [state, value] = stdout.trim().split(/\s+/); + const rss = Number(value); + resolve(state?.startsWith('Z') || !Number.isInteger(rss) || rss <= 0 ? null : rss); + })); +} + +function parseSummary(log) { + const line = log.trim().split('\n').reverse().find((row) => row.startsWith('device=') && row.includes(' frames=')); + if (!line) return null; + const values = {}; + for (const match of line.matchAll(/(?:^| )([a-z0-9_]+)=([^ ]+)/g)) { + const number = Number(match[2]); + values[match[1]] = Number.isFinite(number) ? number : match[2]; + } + return values; +} + +async function main() { + let options; + try { options = optionsOf(process.argv.slice(2)); } + catch (error) { console.error(error.message); usage(); return 2; } + if (options.help) { usage(); return 0; } + if (process.platform !== 'darwin') { console.error('This acceptance runner targets macOS'); return 2; } + if (!fs.existsSync(options.binary) || !fs.existsSync(path.join(path.dirname(options.binary), 'python27.zip')) + || !fs.existsSync(path.join(options.client, 'pack/Index'))) { + console.error('Missing native live-client binary, adjacent python27.zip, or Client/pack/Index'); + return 2; + } + if (fs.existsSync(options.output)) { console.error(`Output already exists: ${options.output}`); return 2; } + let server = null; + if (options.server) { + try { server = parseServer(options.server); } + catch (error) { console.error(error.message); return 2; } + for (const port of [server.authPort, server.gamePort]) { + if (!await reachable(server.host, port)) { + console.error(`Server endpoint unreachable: ${server.host}:${port}; client not started`); + return 2; + } + } + } + fs.mkdirSync(options.output, { recursive: true, mode: 0o700 }); + const logPath = path.join(options.output, 'client.log'); + const rssPath = path.join(options.output, 'rss.jsonl'); + const logFile = fs.openSync(logPath, 'w', 0o600); + const rssFile = fs.openSync(rssPath, 'w', 0o600); + const args = ['--live-client', options.client]; + if (server) args.push('--live-server', options.server, '--login-screen'); + else args.push('--fake-mobs', '24', '--frames', String(options.frames)); + const started = Date.now(); + const child = spawn(options.binary, args, { cwd: repo, stdio: ['inherit', 'pipe', 'pipe'] }); + const literals = secretLiterals(); + let log = ''; + let pending = ''; + const append = (chunk) => { + pending += chunk.toString('utf8'); + const lines = pending.split('\n'); + pending = lines.pop(); + for (const line of lines) { + const clean = `${redact(line, literals)}\n`; + fs.writeSync(logFile, clean); + process.stdout.write(clean); + log += clean; + } + }; + child.stdout.on('data', append); + child.stderr.on('data', append); + let sampleBusy = false; + let pendingSample = Promise.resolve(); + const samples = []; + const sample = async () => { + if (sampleBusy) return; + sampleBusy = true; + const rss = await readRss(child.pid); + if (rss !== null) { + const row = { wall_ms: Date.now(), rss_kib: rss }; + samples.push(row); + fs.writeSync(rssFile, `${JSON.stringify(row)}\n`); + } + sampleBusy = false; + }; + await sample(); + const interval = setInterval(() => { pendingSample = sample(); }, 1000); + const onSignal = (signal) => child.kill(signal); + process.on('SIGINT', onSignal); + process.on('SIGTERM', onSignal); + const result = await new Promise((resolve) => { + child.once('error', (error) => resolve({ code: null, signal: null, error: error.message })); + child.once('close', (code, signal) => resolve({ code, signal })); + }); + clearInterval(interval); + await pendingSample; + process.off('SIGINT', onSignal); + process.off('SIGTERM', onSignal); + if (pending) append('\n'); + fs.closeSync(logFile); + fs.closeSync(rssFile); + const durationSeconds = Math.round((Date.now() - started) / 1000); + const summary = parseSummary(log); + const completed = result.code === 0 && summary?.frames > 0; + const status = !completed ? 'FAIL' : server ? 'NEEDS_MANUAL_REVIEW' : 'SMOKE_PASS'; + const report = { + status, + mode: server ? 'real_server' : 'fake_server', + started_at: new Date(started).toISOString(), duration_seconds: durationSeconds, + minimum_seconds: server ? options.minSeconds : null, + minimum_met: server ? durationSeconds >= options.minSeconds : null, + exit: result, summary, + rss: { samples: samples.length, first_kib: samples[0]?.rss_kib ?? null, + last_kib: samples.at(-1)?.rss_kib ?? null, + max_kib: samples.length ? Math.max(...samples.map((row) => row.rss_kib)) : null }, + manual_checks: server ? { + login: null, character_selection: null, movement: null, combat: null, + map_change: null, inventory_and_chat: null, resize_and_focus: null, + visual_comparison: null, + } : null, + note: server ? 'Review manual checks and frame/RSS trends; this runner does not certify gameplay or memory leaks.' + : 'Fake-server smoke checks startup and rendering only.', + }; + const reportPath = path.join(options.output, 'report.json'); + fs.writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 }); + console.log(`Native acceptance: ${status}; report=${reportPath}`); + return completed ? 0 : 1; +} + +main().then((code) => { process.exitCode = code; }).catch((error) => { + console.error(error.message); + process.exitCode = 1; +}); diff --git a/script/python_game_render_test.sh b/script/python_game_render_test.sh index 62c0776b..d5555afc 100755 --- a/script/python_game_render_test.sh +++ b/script/python_game_render_test.sh @@ -3,6 +3,8 @@ # commands; project/python_game_render_test.gd takes the screenshot and writes report.json. # # script/python_game_render_test.sh offline: port_fake_login_server on loopback +# MT_FAKE_MOB_COUNT=24 MT_PROFILE_FRAME=1 MT_PROFILE_COMBAT=1 script/python_game_render_test.sh +# offline crowd sample (1..64 dogs; default 1) # script/python_game_render_test.sh --live live: MT_LIVE_HOST / MT_LIVE_LOGIN / MT_LIVE_PASSWORD # (+ MT_LIVE_AUTH_PORT / MT_LIVE_GAME_PORT / MT_LIVE_SLOT) # diff --git a/test/run_select_character_test.sh b/test/run_select_character_test.sh new file mode 100755 index 00000000..8e60b2de --- /dev/null +++ b/test/run_select_character_test.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +set -euo pipefail +cd "$(dirname "$0")/.." +repo="$PWD" +godot_bin="${GODOT:-$(which godot 2>/dev/null || echo "/Applications/Godot.app/Contents/MacOS/Godot")}" +server="$repo/build/extension/src/port/port_fake_login_server" + +server_pid="" +cleanup() { + if [ -n "$server_pid" ]; then + kill -TERM "$server_pid" 2>/dev/null || true + wait "$server_pid" 2>/dev/null || true + fi +} +trap cleanup EXIT + +fifo="/tmp/server.ports" +rm -f "$fifo" +mkfifo "$fifo" 2>/dev/null || touch "$fifo" +"$server" > "$fifo" 2> "/tmp/server.log" & +server_pid=$! +for _ in $(seq 50); do + [ -s "$fifo" ] && break + sleep 0.1 +done +read -r word auth_port game_port < "$fifo" || true +if [ "${word:-}" != "ports" ]; then + echo "Error: fake server did not start" >&2 + exit 1 +fi + +export MT_LIVE_HOST=127.0.0.1 +export MT_LIVE_LOGIN=fakeuser +export MT_LIVE_PASSWORD=fakepass +export MT_LIVE_AUTH_PORT="$auth_port" +export MT_LIVE_GAME_PORT="$game_port" +export MT_LIVE_SLOT=0 + +echo "Running test with Godot: $godot_bin" +"$godot_bin" --path "$repo/project" --script res://test_select_character.gd