codepage: replace iconv with bundled WindowsBestFit tables
mt_codepage implements code pages 874/932/936/949/950/1250-1258 from Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py, SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/ IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build no longer depends on iconv. port.codepage checks every table against its source file. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
07c101fdc8
commit
0b701d88c3
+11
-2
@@ -353,7 +353,16 @@ extension/third_party/cpython-2.7.18/ # 静态库(2P 批次
|
||||
| ~~Linux x86_64~~ | 不做(2026-09-23 定,见第 1 节) |
|
||||
| ~~Windows x64~~ | 不做(2026-09-23 定,见第 1 节)。记录原因备查:上游 2.7 只支持 MSVC + `PC/pyconfig.h`,而本仓库的 Windows 门禁是 mingw-w64 交叉编译,`PC/pyconfig.h` 的 gnu-win32 分支不设 `HAVE_UNISTD_H`,`posixmodule.c`/`dynload_win.c` 与 MinGW 头文件冲突。因此 mingw 门禁下不建 `mtpython`,`port_logic`/`port_platform` 自动排除 ScriptLib |
|
||||
|
||||
### 已知问题:Android 的 GBK 编解码(挂起,2026-09-23)
|
||||
### 已解决:Android 的 GBK 编解码(2026-09-28)
|
||||
|
||||
`extension/src/codepage`(`mt_codepage`)按微软 WindowsBestFit 表实现 874/932/936/949/950/1250–1258,
|
||||
`Win32Crt.cpp` 的 `MultiByteToWideChar`/`WideCharToMultiByte`/`IsDBCSLeadByteEx` 和 `net/text_codec.cpp` 都改用它,
|
||||
iconv 已全部移除。表由 `script/gen_codepage_tables.py` 从 unicode.org 下载(SHA-256 固定)生成;
|
||||
`port.codepage` 测试逐条校验 14 张表与源文件一致(解码全部单字节/双字节序列,编码全部 BMP 码位,含/不含 best fit)。
|
||||
与 macOS iconv 一次性逐字对比:CP936 22,111 个码位一致,差异只在 Windows 特有的部分——0x80→€、EUDC 用户造字区映射到
|
||||
私用区(U+E000–F8FF,1,955 个)、A2E3/A8BF 两处;1250/1251/1253/1254 只差未定义字节(Windows 映射为 C1 控制码)。下面是当时的分析记录。
|
||||
|
||||
#### 原记录(2026-09-23)
|
||||
|
||||
**优先级:macOS 先跑通,Android 后面再说**(用户 2026-09-23 定)。
|
||||
|
||||
@@ -515,7 +524,7 @@ extension/third_party/cpython-2.7.18/ # 静态库(2P 批次
|
||||
`Eternexus/root/*.msm` 的路径、大小、sha256,结果在 `audit/assets/40250-client.json`(223 个文件,1,407,474,045
|
||||
字节)。文件缺失、内容变化或清单外多出的文件都会让 verify 失败。
|
||||
|
||||
已知问题(不是本批引入的):完整 Android 链接在 API 24 上失败,原因是 `extension/src/net/text_codec.cpp` 用了 iconv;
|
||||
已知问题(不是本批引入的,2026-09-28 已随 `mt_codepage` 解决):完整 Android 链接在 API 24 上失败,原因是 `extension/src/net/text_codec.cpp` 用了 iconv;
|
||||
本批的 mtproto、`pack40250_node`、`asset_io`、`proto_node` 都能在 Android 上编译。
|
||||
|
||||
## 6. 进度与 port-map 基线
|
||||
|
||||
@@ -14,6 +14,8 @@ add_subdirectory(godot-cpp)
|
||||
# aliases mt3p::sodium / mt3p::zstd / mt3p::minilzo. See docs/THIRD-PARTY.md.
|
||||
add_subdirectory(third_party)
|
||||
|
||||
add_subdirectory(src/codepage)
|
||||
|
||||
# --- mtnet: Metin2 net protocol core (no godot-cpp dep; libsodium only) ---
|
||||
add_library(mtnet STATIC
|
||||
src/net/text_codec.cpp
|
||||
@@ -27,13 +29,7 @@ add_library(mtnet STATIC
|
||||
src/net/classic/classic_cipher.cpp
|
||||
)
|
||||
target_include_directories(mtnet PUBLIC src/net)
|
||||
target_link_libraries(mtnet PUBLIC mt3p::sodium mt3p::minilzo mt3p::cryptopp)
|
||||
# macOS ships iconv as a system library rather than libc. Linux/glibc and
|
||||
# Windows use the platform implementation without this extra link item.
|
||||
find_library(MT_ICONV_LIBRARY NAMES iconv)
|
||||
if(MT_ICONV_LIBRARY)
|
||||
target_link_libraries(mtnet PUBLIC ${MT_ICONV_LIBRARY})
|
||||
endif()
|
||||
target_link_libraries(mtnet PUBLIC mt3p::sodium mt3p::minilzo mt3p::cryptopp mt_codepage)
|
||||
target_compile_features(mtnet PUBLIC cxx_std_20)
|
||||
# §1.5: non-EUROPE CG_CLIENT_VERSION timestamp, in C __TIMESTAMP__ form
|
||||
# ("Www Mmm dd hh:mm:ss yyyy"); regenerated on each CMake configure.
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
# mt_codepage — Windows ANSI code pages (874/932/936/949/950/1250-1258) from Microsoft's WindowsBestFit
|
||||
# tables; codepage_tables.inc is generated by script/gen_codepage_tables.py. Used by port_logic
|
||||
# (Win32Crt MultiByteToWideChar / WideCharToMultiByte) and mtnet (40250 CP936 wire text), in place of
|
||||
# the host iconv.
|
||||
add_library(mt_codepage STATIC codepage.cpp)
|
||||
target_include_directories(mt_codepage PUBLIC ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
target_compile_features(mt_codepage PUBLIC cxx_std_17)
|
||||
set_target_properties(mt_codepage PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
@@ -0,0 +1,169 @@
|
||||
#include "codepage.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
|
||||
namespace mt_codepage {
|
||||
namespace {
|
||||
|
||||
constexpr uint16_t kNone = 0xFFFF;
|
||||
|
||||
struct DbcsRow {
|
||||
uint8_t lead;
|
||||
uint8_t trail_lo;
|
||||
uint8_t trail_hi;
|
||||
uint32_t offset; // into the Dbcs array, trail_lo first
|
||||
};
|
||||
|
||||
// A WCTABLE entry the first-match inverse of the byte tables gets wrong.
|
||||
struct EncodeOverride {
|
||||
uint16_t wc;
|
||||
uint16_t mb;
|
||||
uint8_t kind; // 0 exact (round-trips), 1 best fit only, 2 not encodable
|
||||
};
|
||||
|
||||
struct CodePageTable {
|
||||
unsigned cp;
|
||||
uint16_t mb_default; // Unicode char for an unmapped byte / pair (CPINFO)
|
||||
const uint16_t* single;
|
||||
const uint8_t* leads; // lo, hi pairs
|
||||
size_t lead_count;
|
||||
const DbcsRow* rows;
|
||||
size_t row_count;
|
||||
const uint16_t* dbcs;
|
||||
const EncodeOverride* overrides;
|
||||
size_t override_count;
|
||||
};
|
||||
|
||||
#include "codepage_tables.inc"
|
||||
|
||||
constexpr size_t kTableCount = sizeof(kTables) / sizeof(kTables[0]);
|
||||
|
||||
const CodePageTable* find(unsigned cp)
|
||||
{
|
||||
for (const CodePageTable& t : kTables)
|
||||
if (t.cp == cp) return &t;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool lead(const CodePageTable& t, unsigned char b)
|
||||
{
|
||||
for (size_t i = 0; i < t.lead_count; ++i)
|
||||
if (b >= t.leads[i * 2] && b <= t.leads[i * 2 + 1]) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t pair(const CodePageTable& t, unsigned char l, unsigned char trail)
|
||||
{
|
||||
for (size_t i = 0; i < t.row_count; ++i) {
|
||||
const DbcsRow& r = t.rows[i];
|
||||
if (r.lead != l) continue;
|
||||
if (trail < r.trail_lo || trail > r.trail_hi) return kNone;
|
||||
return t.dbcs[r.offset + (trail - r.trail_lo)];
|
||||
}
|
||||
return kNone;
|
||||
}
|
||||
|
||||
// Exact encoder (BMP code point -> bytes, kNone when unmapped), built on first use per code page.
|
||||
struct Encoder {
|
||||
std::once_flag once;
|
||||
std::unique_ptr<uint16_t[]> exact;
|
||||
};
|
||||
Encoder g_encoders[kTableCount];
|
||||
|
||||
const uint16_t* exact_encoder(const CodePageTable& t)
|
||||
{
|
||||
Encoder& e = g_encoders[&t - kTables];
|
||||
std::call_once(e.once, [&] {
|
||||
e.exact.reset(new uint16_t[0x10000]);
|
||||
std::fill_n(e.exact.get(), 0x10000, kNone);
|
||||
uint16_t* map = e.exact.get();
|
||||
// First match in ascending byte order, as the generator assumed.
|
||||
for (unsigned b = 0; b < 256; ++b)
|
||||
if (t.single[b] != kNone && map[t.single[b]] == kNone) map[t.single[b]] = uint16_t(b);
|
||||
for (size_t i = 0; i < t.row_count; ++i) {
|
||||
const DbcsRow& r = t.rows[i];
|
||||
for (unsigned trail = r.trail_lo; trail <= r.trail_hi; ++trail) {
|
||||
const uint16_t wc = t.dbcs[r.offset + (trail - r.trail_lo)];
|
||||
if (wc != kNone && map[wc] == kNone) map[wc] = uint16_t(r.lead << 8 | trail);
|
||||
}
|
||||
}
|
||||
for (size_t i = 0; i < t.override_count; ++i) {
|
||||
const EncodeOverride& o = t.overrides[i];
|
||||
if (o.kind == 0) map[o.wc] = o.mb;
|
||||
else if (o.kind == 2) map[o.wc] = kNone;
|
||||
}
|
||||
});
|
||||
return e.exact.get();
|
||||
}
|
||||
|
||||
uint16_t best_fit(const CodePageTable& t, char32_t c)
|
||||
{
|
||||
const EncodeOverride* end = t.overrides + t.override_count;
|
||||
const EncodeOverride* it = std::lower_bound(t.overrides, end, c,
|
||||
[](const EncodeOverride& o, char32_t v) { return o.wc < v; });
|
||||
return it != end && it->wc == c && it->kind == 1 ? it->mb : kNone;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool supported(unsigned cp)
|
||||
{
|
||||
return find(cp) != nullptr;
|
||||
}
|
||||
|
||||
bool is_lead_byte(unsigned cp, unsigned char b)
|
||||
{
|
||||
const CodePageTable* t = find(cp);
|
||||
return t && lead(*t, b);
|
||||
}
|
||||
|
||||
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
const CodePageTable* t = find(cp);
|
||||
if (!t) return false;
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < n;) {
|
||||
const unsigned char b = p[i];
|
||||
uint16_t wc = kNone;
|
||||
if (lead(*t, b) && i + 1 < n && p[i + 1] != 0) {
|
||||
wc = pair(*t, b, p[i + 1]);
|
||||
i += 2;
|
||||
} else {
|
||||
wc = t->single[b];
|
||||
++i;
|
||||
}
|
||||
if (wc == kNone) {
|
||||
wc = t->mb_default;
|
||||
ok = false;
|
||||
}
|
||||
out.push_back(wc);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
void encode(unsigned cp, std::u32string_view in, std::string& out, bool allow_best_fit, bool* used_default)
|
||||
{
|
||||
const CodePageTable* t = find(cp);
|
||||
const uint16_t* exact = t ? exact_encoder(*t) : nullptr;
|
||||
for (char32_t c : in) {
|
||||
uint16_t mb = kNone;
|
||||
if (exact && c < 0x10000) {
|
||||
mb = exact[c];
|
||||
if (mb == kNone && allow_best_fit) mb = best_fit(*t, c);
|
||||
}
|
||||
if (mb == kNone) {
|
||||
out.push_back('?');
|
||||
if (used_default) *used_default = true;
|
||||
} else if (mb > 0xFF) {
|
||||
out.push_back(char(mb >> 8));
|
||||
out.push_back(char(mb & 0xFF));
|
||||
} else {
|
||||
out.push_back(char(mb));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mt_codepage
|
||||
@@ -0,0 +1,29 @@
|
||||
#pragma once
|
||||
// Windows ANSI code pages without the host iconv: MultiByteToWideChar / WideCharToMultiByte semantics
|
||||
// driven by Microsoft's WindowsBestFit tables (script/gen_codepage_tables.py). One implementation for
|
||||
// macOS, Linux, Android and iOS, so every platform decodes the 40250 locale and wire text the same way.
|
||||
//
|
||||
// Covered: 874, 932, 936, 949, 950, 1250-1258 (every code page 40250 EterLib/Util.cpp names).
|
||||
// UTF-8 is not handled here.
|
||||
|
||||
#include <cstddef>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
|
||||
namespace mt_codepage {
|
||||
|
||||
bool supported(unsigned cp);
|
||||
|
||||
// Byte string -> code points, as MultiByteToWideChar. A lead byte takes the next non-NUL byte as its
|
||||
// trail; an unmapped byte or pair becomes the code page's default char (U+30FB for 932, else '?').
|
||||
// Returns false when a default char was substituted (MB_ERR_INVALID_CHARS would fail).
|
||||
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out);
|
||||
|
||||
// Code points -> bytes, as WideCharToMultiByte. With best_fit (no WC_NO_BEST_FIT_CHARS) a character
|
||||
// without an exact mapping may take its WCTABLE best fit ("Ā" -> "A" on 1252); otherwise, and for
|
||||
// anything unmapped, '?' is written and *used_default set.
|
||||
void encode(unsigned cp, std::u32string_view in, std::string& out, bool best_fit, bool* used_default);
|
||||
|
||||
bool is_lead_byte(unsigned cp, unsigned char b);
|
||||
|
||||
} // namespace mt_codepage
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,16 +1,12 @@
|
||||
#include "text_codec.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cerrno>
|
||||
#include <cstring>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include <windows.h>
|
||||
#elif defined(__has_include)
|
||||
#if __has_include(<iconv.h>)
|
||||
#include <iconv.h>
|
||||
#define MT_TEXT_CODEC_HAS_ICONV 1
|
||||
#endif
|
||||
#else
|
||||
// CP936 from Microsoft's own table, not the host iconv (bionic has no GBK).
|
||||
#include "codepage.h"
|
||||
#endif
|
||||
|
||||
namespace mtnet {
|
||||
@@ -69,61 +65,66 @@ bool convert_from_windows(std::string_view input, std::string &out) {
|
||||
return WideCharToMultiByte(CP_UTF8, 0, wide.data(), wide_len, out.data(), utf8_len,
|
||||
nullptr, nullptr) == utf8_len;
|
||||
}
|
||||
#elif defined(MT_TEXT_CODEC_HAS_ICONV)
|
||||
bool convert_iconv(std::string_view input, const char *from, const char *to, std::string &out) {
|
||||
iconv_t cd = iconv_open(to, from);
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
std::string buffer(std::max<size_t>(input.size() * 4 + 16, 32), '\0');
|
||||
char *in_ptr = const_cast<char *>(input.data());
|
||||
size_t in_left = input.size();
|
||||
char *out_ptr = buffer.data();
|
||||
size_t out_left = buffer.size();
|
||||
while (in_left != 0) {
|
||||
const size_t rc = iconv(cd, &in_ptr, &in_left, &out_ptr, &out_left);
|
||||
if (rc == static_cast<size_t>(-1)) {
|
||||
iconv_close(cd);
|
||||
#else
|
||||
// Mirrors the Windows branch: CP936 without best-fit substitutions, and a strict decode.
|
||||
constexpr unsigned kWireCodePage = 936;
|
||||
|
||||
bool decode_utf8_strict(std::string_view input, std::u32string &out) {
|
||||
for (size_t i = 0; i < input.size();) {
|
||||
const size_t seq = utf8_sequence_length(input.data() + i, input.size() - i);
|
||||
if (seq == 0) return false;
|
||||
const unsigned char c = static_cast<unsigned char>(input[i]);
|
||||
char32_t cp = seq == 1 ? c : seq == 2 ? (c & 0x1F) : seq == 3 ? (c & 0x0F) : (c & 0x07);
|
||||
for (size_t k = 1; k < seq; ++k) {
|
||||
const unsigned char t = static_cast<unsigned char>(input[i + k]);
|
||||
if ((t & 0xC0) != 0x80) return false;
|
||||
cp = (cp << 6) | (t & 0x3F);
|
||||
}
|
||||
if ((seq == 3 && cp < 0x800) || (seq == 4 && (cp < 0x10000 || cp > 0x10FFFF)) ||
|
||||
(cp >= 0xD800 && cp <= 0xDFFF)) {
|
||||
return false;
|
||||
}
|
||||
out.push_back(cp);
|
||||
i += seq;
|
||||
}
|
||||
// Flush stateful encodings, even though GB2312 itself is stateless.
|
||||
iconv(cd, nullptr, nullptr, &out_ptr, &out_left);
|
||||
iconv_close(cd);
|
||||
out.assign(buffer.data(), static_cast<size_t>(out_ptr - buffer.data()));
|
||||
return true;
|
||||
}
|
||||
|
||||
void append_utf8(char32_t cp, std::string &out) {
|
||||
if (cp < 0x80) {
|
||||
out.push_back(static_cast<char>(cp));
|
||||
} else if (cp < 0x800) {
|
||||
out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
|
||||
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
||||
} else if (cp < 0x10000) {
|
||||
out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
|
||||
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
|
||||
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
||||
} else {
|
||||
out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
|
||||
out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
|
||||
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
|
||||
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
||||
}
|
||||
}
|
||||
|
||||
bool encode_one(std::string_view input, std::string &out) {
|
||||
return convert_iconv(input, "UTF-8", "GB2312", out) ||
|
||||
convert_iconv(input, "UTF-8", "GBK", out);
|
||||
std::u32string wide;
|
||||
if (!decode_utf8_strict(input, wide)) return false;
|
||||
bool used_default = false;
|
||||
out.clear();
|
||||
mt_codepage::encode(kWireCodePage, wide, out, false, &used_default);
|
||||
return !used_default;
|
||||
}
|
||||
|
||||
bool decode_wire(std::string_view input, std::string &out) {
|
||||
return convert_iconv(input, "GB2312", "UTF-8", out) ||
|
||||
convert_iconv(input, "GBK", "UTF-8", out);
|
||||
}
|
||||
#else
|
||||
bool encode_one(std::string_view input, std::string &out) {
|
||||
// A platform without either Windows code-page APIs or iconv must never
|
||||
// send UTF-8 bytes into a 40250 fixed legacy field. ASCII is lossless;
|
||||
// non-ASCII is reported as unrepresentable so wire_text_fits() rejects it
|
||||
// and to_wire() can safely substitute '?' rather than leaking a 3-byte
|
||||
// UTF-8 sequence into a 24-byte name field.
|
||||
const size_t n = utf8_byte_truncate(input.data(), input.size(), input.size());
|
||||
if (n != input.size()) return false;
|
||||
for (unsigned char c : input) {
|
||||
if (c >= 0x80) return false;
|
||||
std::u32string wide;
|
||||
if (!mt_codepage::decode(kWireCodePage, reinterpret_cast<const unsigned char *>(input.data()),
|
||||
input.size(), wide)) {
|
||||
return false;
|
||||
}
|
||||
out.assign(input.data(), input.size());
|
||||
return true;
|
||||
}
|
||||
|
||||
bool decode_wire(std::string_view input, std::string &out) {
|
||||
// The fallback decoder is deliberately conservative for the same reason:
|
||||
// returning raw high-bit bytes would create invalid UTF-8 in the client.
|
||||
for (unsigned char c : input) {
|
||||
if (c >= 0x80) return false;
|
||||
}
|
||||
out.assign(input.data(), input.size());
|
||||
out.clear();
|
||||
for (char32_t c : wide) append_utf8(c, out);
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
@@ -176,7 +177,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
|
||||
size_t n = 0;
|
||||
while (n < cap && bytes[n] != '\0') ++n;
|
||||
std::string decoded;
|
||||
#if defined(_WIN32) || defined(MT_TEXT_CODEC_HAS_ICONV)
|
||||
// A fixed field can arrive without a terminator and a caller may provide a
|
||||
// larger view than the actual field. If that view ends on a legacy lead byte,
|
||||
// drop only the malformed suffix and preserve the complete preceding text.
|
||||
@@ -188,11 +188,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
|
||||
#endif
|
||||
if (ok) return decoded;
|
||||
}
|
||||
#else
|
||||
if (decode_wire(std::string_view(bytes, n), decoded)) {
|
||||
return decoded;
|
||||
}
|
||||
#endif
|
||||
// Keep malformed input visible without leaking legacy bytes as invalid UTF-8.
|
||||
decoded.clear();
|
||||
for (size_t i = 0; i < n; ++i) {
|
||||
|
||||
@@ -31,12 +31,12 @@ target_include_directories(port_logic PUBLIC ${MT_PORT_SHIMS})
|
||||
# Libraries 40250 links that are vendored as-is: <lzo/lzo1x.h> (EterBase/lzo.h), <cryptopp/*>
|
||||
# (EterBase/cipher.h), <Python-2.7/*> (ScriptLib).
|
||||
target_link_libraries(port_logic PUBLIC mt3p::minilzo mt3p::cryptopp)
|
||||
# common/Win32Crt.cpp converts non-UTF-8 code pages with iconv. macOS ships it as a separate library;
|
||||
# extension/CMakeLists.txt finds it for mtnet, but the port-only build (script/port_gate.sh) never reads that file.
|
||||
find_library(MT_ICONV_LIBRARY NAMES iconv)
|
||||
if(MT_ICONV_LIBRARY)
|
||||
target_link_libraries(port_logic PUBLIC ${MT_ICONV_LIBRARY})
|
||||
# common/Win32Crt.cpp converts ANSI code pages with mt_codepage. extension/CMakeLists.txt adds it for
|
||||
# mtnet; the port-only build (script/port_gate.sh, native_render) never reads that file.
|
||||
if(NOT TARGET mt_codepage)
|
||||
add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/../codepage ${CMAKE_CURRENT_BINARY_DIR}/../codepage)
|
||||
endif()
|
||||
target_link_libraries(port_logic PUBLIC mt_codepage)
|
||||
if(TARGET mtpython)
|
||||
target_link_libraries(port_logic PUBLIC mt3p::python)
|
||||
endif()
|
||||
@@ -173,6 +173,14 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
|
||||
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})
|
||||
set_tests_properties(port.eterpack PROPERTIES SKIP_RETURN_CODE 77)
|
||||
|
||||
# mt_codepage against Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py caches them
|
||||
# in build/bestfit); skipped (77) without them unless MT_ASSETS_STRICT=1.
|
||||
add_executable(port_codepage_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/codepage_test.cpp)
|
||||
target_link_libraries(port_codepage_test PRIVATE mt_codepage)
|
||||
add_test(NAME port.codepage COMMAND $<TARGET_FILE:port_codepage_test>
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../../../build/bestfit)
|
||||
set_tests_properties(port.codepage PROPERTIES SKIP_RETURN_CODE 77)
|
||||
|
||||
# 批次 2V0-g: the 40250 font path over the FreeType GDI emulation; needs a system Tahoma.
|
||||
add_executable(port_text_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/port_text_test.cpp)
|
||||
target_link_libraries(port_text_test PRIVATE port_platform)
|
||||
|
||||
@@ -11,11 +11,11 @@
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
#if __has_include(<iconv.h>)
|
||||
#include <iconv.h>
|
||||
#define MT_WIN32CRT_HAS_ICONV 1
|
||||
#endif
|
||||
#include <glob.h>
|
||||
// ANSI code pages come from Microsoft's own tables, not the host iconv (bionic has no GBK).
|
||||
#include "codepage.h"
|
||||
#include <algorithm>
|
||||
#include <dirent.h>
|
||||
#include <fnmatch.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
@@ -103,17 +103,30 @@ HANDLE FindFirstFile(const char* pattern, WIN32_FIND_DATA* data)
|
||||
std::string normalized(pattern);
|
||||
for (char& ch : normalized)
|
||||
if (ch == '\\') ch = '/';
|
||||
glob_t matches{};
|
||||
const int result = glob(normalized.c_str(), 0, nullptr, &matches);
|
||||
if (result != 0 || matches.gl_pathc == 0)
|
||||
// Win32 wildcards only appear in the last component; match it against the directory like
|
||||
// glob(3) would (sorted, '*' skips dot files). glob itself is API 28+ on Android.
|
||||
std::vector<std::string> paths;
|
||||
const size_t slash = normalized.rfind('/');
|
||||
const std::string dir = slash == std::string::npos ? std::string() : normalized.substr(0, slash + 1);
|
||||
const std::string name = normalized.substr(dir.size());
|
||||
if (name.find_first_of("*?[") == std::string::npos)
|
||||
{
|
||||
globfree(&matches);
|
||||
return INVALID_HANDLE_VALUE;
|
||||
struct stat st;
|
||||
if (::stat(normalized.c_str(), &st) == 0)
|
||||
paths.push_back(normalized);
|
||||
}
|
||||
else if (DIR* d = ::opendir(dir.empty() ? "." : dir.c_str()))
|
||||
{
|
||||
while (const dirent* entry = ::readdir(d))
|
||||
if (::fnmatch(name.c_str(), entry->d_name, FNM_PERIOD) == 0)
|
||||
paths.push_back(dir + entry->d_name);
|
||||
::closedir(d);
|
||||
std::sort(paths.begin(), paths.end());
|
||||
}
|
||||
if (paths.empty())
|
||||
return INVALID_HANDLE_VALUE;
|
||||
auto* state = new FindState;
|
||||
for (size_t i = 0; i < matches.gl_pathc; ++i)
|
||||
state->paths.emplace_back(matches.gl_pathv[i]);
|
||||
globfree(&matches);
|
||||
state->paths = std::move(paths);
|
||||
fill_find_data(state->paths[0], data);
|
||||
state->next = 1;
|
||||
return state;
|
||||
@@ -199,13 +212,8 @@ DWORD GetTickCount()
|
||||
|
||||
namespace {
|
||||
|
||||
// Windows-1252 0x80..0x9F; the other bytes map to the same code point (0 marks the undefined ones).
|
||||
const char32_t kCp1252High[32] = {
|
||||
0x20AC, 0, 0x201A, 0x0192, 0x201E, 0x2026, 0x2020, 0x2021, 0x02C6, 0x2030, 0x0160, 0x2039, 0x0152, 0, 0x017D, 0,
|
||||
0, 0x2018, 0x2019, 0x201C, 0x201D, 0x2022, 0x2013, 0x2014, 0x02DC, 0x2122, 0x0161, 0x203A, 0x0153, 0, 0x017E, 0x0178,
|
||||
};
|
||||
|
||||
bool is_cp1252(UINT cp) { return cp == CP_ACP || cp == 1252; }
|
||||
// CP_ACP is the Western system code page here, as on the 40250 target machines.
|
||||
UINT ansi(UINT cp) { return cp == CP_ACP ? 1252 : cp; }
|
||||
|
||||
bool decode_utf8(const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
@@ -237,108 +245,25 @@ void encode_utf8(char32_t cp, std::string& out)
|
||||
else { out.push_back(char(0xF0 | (cp >> 18))); out.push_back(char(0x80 | ((cp >> 12) & 0x3F))); out.push_back(char(0x80 | ((cp >> 6) & 0x3F))); out.push_back(char(0x80 | (cp & 0x3F))); }
|
||||
}
|
||||
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
std::string iconv_name(UINT cp)
|
||||
{
|
||||
switch (cp) {
|
||||
case 949: return "CP949";
|
||||
case 936: return "GBK";
|
||||
case 950: return "BIG5";
|
||||
case 932: return "SHIFT_JIS";
|
||||
case 874: return "CP874";
|
||||
default: return "CP" + std::to_string(cp);
|
||||
}
|
||||
}
|
||||
|
||||
// One character at a time, so an undecodable byte becomes one default char like Windows does.
|
||||
bool iconv_decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
iconv_t cd = iconv_open("UTF-32LE", iconv_name(cp).c_str());
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
for (size_t i = 0; i < n;) {
|
||||
unsigned char wide[16];
|
||||
bool done = false;
|
||||
for (size_t len = 1; len <= 4 && i + len <= n && !done; ++len) {
|
||||
iconv(cd, nullptr, nullptr, nullptr, nullptr);
|
||||
char* in = const_cast<char*>(reinterpret_cast<const char*>(p + i));
|
||||
size_t in_left = len;
|
||||
char* o = reinterpret_cast<char*>(wide);
|
||||
size_t o_left = sizeof(wide);
|
||||
if (iconv(cd, &in, &in_left, &o, &o_left) != size_t(-1) && in_left == 0 && o_left < sizeof(wide)) {
|
||||
for (size_t k = 0; k + 4 <= sizeof(wide) - o_left; k += 4)
|
||||
out.push_back(char32_t(wide[k] | (wide[k + 1] << 8) | (wide[k + 2] << 16) | (unsigned(wide[k + 3]) << 24)));
|
||||
i += len;
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
if (!done) { out.push_back(0xFFFD); ++i; }
|
||||
}
|
||||
iconv_close(cd);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool iconv_encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
|
||||
{
|
||||
iconv_t cd = iconv_open(iconv_name(cp).c_str(), "UTF-32LE");
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
for (char32_t c : in) {
|
||||
unsigned char src[4] = {static_cast<unsigned char>(c), static_cast<unsigned char>(c >> 8), static_cast<unsigned char>(c >> 16), static_cast<unsigned char>(c >> 24)};
|
||||
char buf[16];
|
||||
char* ip = reinterpret_cast<char*>(src);
|
||||
size_t il = 4;
|
||||
char* op = buf;
|
||||
size_t ol = sizeof(buf);
|
||||
iconv(cd, nullptr, nullptr, nullptr, nullptr);
|
||||
if (iconv(cd, &ip, &il, &op, &ol) != size_t(-1) && il == 0) {
|
||||
out.append(buf, sizeof(buf) - ol);
|
||||
} else {
|
||||
out.push_back('?');
|
||||
*used_default = true;
|
||||
}
|
||||
}
|
||||
iconv_close(cd);
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
bool decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
if (cp == CP_UTF8) return decode_utf8(p, n, out);
|
||||
if (is_cp1252(cp)) {
|
||||
for (size_t i = 0; i < n; ++i) {
|
||||
const unsigned char c = p[i];
|
||||
out.push_back(c >= 0x80 && c < 0xA0 ? (kCp1252High[c - 0x80] ? kCp1252High[c - 0x80] : char32_t(c)) : char32_t(c));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
if (iconv_decode(cp, p, n, out)) return true;
|
||||
#endif
|
||||
// No converter for this code page: ASCII passes, the rest becomes U+FFFD.
|
||||
if (mt_codepage::supported(ansi(cp))) return mt_codepage::decode(ansi(cp), p, n, out);
|
||||
// No table for this code page: ASCII passes, the rest becomes U+FFFD.
|
||||
for (size_t i = 0; i < n; ++i) out.push_back(p[i] < 0x80 ? char32_t(p[i]) : char32_t(0xFFFD));
|
||||
return true;
|
||||
}
|
||||
|
||||
void encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
|
||||
void encode(UINT cp, DWORD flags, const std::u32string& in, std::string& out, bool* used_default)
|
||||
{
|
||||
if (cp == CP_UTF8) {
|
||||
for (char32_t c : in) encode_utf8(c, out);
|
||||
return;
|
||||
}
|
||||
if (is_cp1252(cp)) {
|
||||
for (char32_t c : in) {
|
||||
if (c < 0x80 || (c >= 0xA0 && c < 0x100)) { out.push_back(char(c)); continue; }
|
||||
int hit = -1;
|
||||
for (int k = 0; k < 32; ++k)
|
||||
if (kCp1252High[k] == c) { hit = k; break; }
|
||||
if (hit >= 0) out.push_back(char(0x80 + hit));
|
||||
else { out.push_back('?'); *used_default = true; }
|
||||
}
|
||||
if (mt_codepage::supported(ansi(cp))) {
|
||||
mt_codepage::encode(ansi(cp), in, out, (flags & WC_NO_BEST_FIT_CHARS) == 0, used_default);
|
||||
return;
|
||||
}
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
if (iconv_encode(cp, in, out, used_default)) return;
|
||||
#endif
|
||||
for (char32_t c : in) {
|
||||
if (c < 0x80) out.push_back(char(c));
|
||||
else { out.push_back('?'); *used_default = true; }
|
||||
@@ -361,7 +286,7 @@ int MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int
|
||||
return int(wide.size());
|
||||
}
|
||||
|
||||
int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWideChar,
|
||||
int WideCharToMultiByte(UINT CodePage, DWORD dwFlags, LPCWSTR lpWideCharStr, int cchWideChar,
|
||||
LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar, LPBOOL lpUsedDefaultChar)
|
||||
{
|
||||
if (!lpWideCharStr || cchWideChar == 0 || cbMultiByte < 0) return 0;
|
||||
@@ -372,14 +297,14 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
|
||||
for (size_t i = 0; i < n; ++i) wide[i] = char32_t(lpWideCharStr[i]);
|
||||
std::string bytes;
|
||||
bool used_default = false;
|
||||
encode(CodePage, wide, bytes, &used_default);
|
||||
encode(CodePage, dwFlags, wide, bytes, &used_default);
|
||||
if (used_default && lpDefaultChar && CodePage != CP_UTF8) {
|
||||
// '?' was the placeholder; the caller's default char replaces it only where a fallback happened.
|
||||
std::string redo;
|
||||
for (char32_t c : wide) {
|
||||
std::string one;
|
||||
bool miss = false;
|
||||
encode(CodePage, std::u32string(1, c), one, &miss);
|
||||
encode(CodePage, dwFlags, std::u32string(1, c), one, &miss);
|
||||
redo += miss ? std::string(1, *lpDefaultChar) : one;
|
||||
}
|
||||
bytes.swap(redo);
|
||||
@@ -393,13 +318,7 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
|
||||
|
||||
BOOL IsDBCSLeadByteEx(UINT CodePage, BYTE TestChar)
|
||||
{
|
||||
switch (CodePage) {
|
||||
case 932: return (TestChar >= 0x81 && TestChar <= 0x9F) || (TestChar >= 0xE0 && TestChar <= 0xFC);
|
||||
case 936:
|
||||
case 949:
|
||||
case 950: return TestChar >= 0x81 && TestChar <= 0xFE;
|
||||
default: return FALSE;
|
||||
}
|
||||
return mt_codepage::is_lead_byte(ansi(CodePage), TestChar) ? TRUE : FALSE;
|
||||
}
|
||||
|
||||
LPSTR CharNextExA(WORD CodePage, LPCSTR lpCurrentChar, DWORD)
|
||||
|
||||
@@ -104,8 +104,8 @@ void DeleteCriticalSection(LPCRITICAL_SECTION cs);
|
||||
void EnterCriticalSection(LPCRITICAL_SECTION cs);
|
||||
void LeaveCriticalSection(LPCRITICAL_SECTION cs);
|
||||
|
||||
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 and
|
||||
// CP_ACP/1252 are converted directly; other code pages go through iconv where the platform has it.
|
||||
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 is converted
|
||||
// directly; CP_ACP (= 1252) and the other ANSI code pages use Microsoft's tables (codepage/codepage.h).
|
||||
#define CP_ACP 0
|
||||
#define CP_UTF8 65001
|
||||
#define MB_PRECOMPOSED 0x00000001
|
||||
|
||||
@@ -0,0 +1,214 @@
|
||||
// codepage_test — mt_codepage against Microsoft's WindowsBestFit source tables, plus spot checks of the
|
||||
// characters each shipped 40250 locale needs.
|
||||
//
|
||||
// Argument: the directory holding bestfit<cp>.txt (script/gen_codepage_tables.py downloads them to
|
||||
// build/bestfit). Without it only the spot checks run and the test exits 77 (ctest SKIP), unless
|
||||
// MT_ASSETS_STRICT=1.
|
||||
#include "codepage.h"
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
static int g_failures = 0;
|
||||
#define CHECK(cond) \
|
||||
do { \
|
||||
if (!(cond)) { \
|
||||
std::fprintf(stderr, "%s:%d: CHECK(%s)\n", __FILE__, __LINE__, #cond); \
|
||||
++g_failures; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
static std::u32string dec(unsigned cp, const char* bytes, bool* ok = nullptr)
|
||||
{
|
||||
std::u32string out;
|
||||
const bool r = mt_codepage::decode(cp, reinterpret_cast<const unsigned char*>(bytes), std::strlen(bytes), out);
|
||||
if (ok) *ok = r;
|
||||
return out;
|
||||
}
|
||||
|
||||
static std::string enc(unsigned cp, std::u32string in, bool best_fit = false, bool* used_default = nullptr)
|
||||
{
|
||||
std::string out;
|
||||
bool used = false;
|
||||
mt_codepage::encode(cp, in, out, best_fit, &used);
|
||||
if (used_default) *used_default = used;
|
||||
return out;
|
||||
}
|
||||
|
||||
static void spot_checks()
|
||||
{
|
||||
// zh (936): "中文" D6D0 CEC4, and the Windows-only single byte 0x80 = euro.
|
||||
CHECK(dec(936, "\xd6\xd0\xce\xc4") == U"中文");
|
||||
CHECK(enc(936, U"中文") == "\xd6\xd0\xce\xc4");
|
||||
CHECK(dec(936, "\x80") == U"€");
|
||||
// GBK extension beyond GB2312 (U+4E02 = 81 40), both directions.
|
||||
CHECK(dec(936, "\x81\x40") == U"丂");
|
||||
CHECK(enc(936, U"丂") == "\x81\x40");
|
||||
// A lead byte with nothing after it is one default char and reports failure.
|
||||
bool ok = true;
|
||||
CHECK(dec(936, "a\xd6", &ok) == U"a?" && !ok);
|
||||
// An unmapped pair is consumed whole (932 uses U+30FB as its default).
|
||||
CHECK(dec(932, "\x85\x40") == U"・");
|
||||
CHECK(dec(932, "\x82\xa0") == U"あ");
|
||||
|
||||
// The shipped European locales.
|
||||
CHECK(dec(1252, "\x80\xe9") == U"€é"); // en/de/fr/...: euro, é
|
||||
CHECK(dec(1252, "\x81") == U"\u0081"); // undefined byte keeps its value, as Windows
|
||||
CHECK(dec(1250, "\xb9\xb3\x9c") == U"ąłś"); // pl/cz/hu/ro: ą ł ś
|
||||
CHECK(dec(1251, "\xc6\xe8") == U"Жи"); // ru: Жи
|
||||
CHECK(dec(1253, "\xd9\xe1") == U"Ωα"); // gr: Ωα
|
||||
CHECK(dec(1254, "\xf0\xfe\xdd") == U"ğşİ"); // tr: ğ ş İ
|
||||
CHECK(enc(1250, U"ąłś") == "\xb9\xb3\x9c");
|
||||
CHECK(enc(1251, U"Жи") == "\xc6\xe8");
|
||||
CHECK(enc(1253, U"Ωα") == "\xd9\xe1");
|
||||
CHECK(enc(1254, U"ğşİ") == "\xf0\xfe\xdd");
|
||||
|
||||
// Best fit only when WC_NO_BEST_FIT_CHARS is absent: 1252 has no "Ā", its best fit is "A".
|
||||
bool used = false;
|
||||
CHECK(enc(1252, U"Ā", true, &used) == "A" && !used);
|
||||
CHECK(enc(1252, U"Ā", false, &used) == "?" && used);
|
||||
// Unmapped either way; beyond the BMP too.
|
||||
CHECK(enc(1251, U"中", true, &used) == "?" && used);
|
||||
CHECK(enc(936, U"\U0001F600", true, &used) == "?" && used);
|
||||
|
||||
CHECK(mt_codepage::is_lead_byte(936, 0x81) && !mt_codepage::is_lead_byte(936, 0x80));
|
||||
CHECK(mt_codepage::is_lead_byte(932, 0xe0) && !mt_codepage::is_lead_byte(932, 0xa1));
|
||||
CHECK(!mt_codepage::is_lead_byte(1250, 0xb9));
|
||||
CHECK(!mt_codepage::supported(65001) && !mt_codepage::supported(437));
|
||||
}
|
||||
|
||||
struct SourceTable {
|
||||
unsigned mb_default = 0x3f;
|
||||
std::map<unsigned, unsigned> decode; // bytes (pair = lead << 8 | trail) -> wc
|
||||
std::map<unsigned, unsigned> wctable;
|
||||
std::vector<std::pair<unsigned, unsigned>> leads;
|
||||
};
|
||||
|
||||
static bool load(const std::string& path, SourceTable& t)
|
||||
{
|
||||
std::ifstream in(path, std::ios::binary);
|
||||
if (!in) return false;
|
||||
std::string raw, section;
|
||||
unsigned lead = 0;
|
||||
while (std::getline(in, raw)) {
|
||||
std::string line = raw.substr(0, raw.find(';'));
|
||||
std::istringstream words(line);
|
||||
std::string head;
|
||||
if (!(words >> head)) continue;
|
||||
if (head == "CPINFO") {
|
||||
std::string size, byte_default, wc_default;
|
||||
words >> size >> byte_default >> wc_default;
|
||||
t.mb_default = unsigned(std::stoul(wc_default, nullptr, 16));
|
||||
continue;
|
||||
}
|
||||
if (head == "MBTABLE" || head == "WCTABLE" || head == "DBCSRANGE") { section = head; continue; }
|
||||
if (head == "DBCSTABLE") {
|
||||
section = head;
|
||||
lead = unsigned(std::stoul(raw.substr(raw.find("0x", raw.find("LeadByte"))), nullptr, 16));
|
||||
continue;
|
||||
}
|
||||
if (head == "ENDCODEPAGE") break;
|
||||
if (head.rfind("0x", 0) != 0) continue;
|
||||
std::string second;
|
||||
words >> second;
|
||||
const unsigned a = unsigned(std::stoul(head, nullptr, 16));
|
||||
const unsigned b = unsigned(std::stoul(second, nullptr, 16));
|
||||
if (raw.find("Lead Byte Range") != std::string::npos) t.leads.push_back({a, b});
|
||||
else if (section == "MBTABLE") t.decode[a] = b;
|
||||
else if (section == "DBCSTABLE") t.decode[lead << 8 | a] = b;
|
||||
else if (section == "WCTABLE") t.wctable[a] = b;
|
||||
}
|
||||
return !t.decode.empty() && !t.wctable.empty();
|
||||
}
|
||||
|
||||
static std::string bytes_of(unsigned mb)
|
||||
{
|
||||
return mb > 0xff ? std::string{char(mb >> 8), char(mb & 0xff)} : std::string(1, char(mb));
|
||||
}
|
||||
|
||||
// Every byte / pair the source defines decodes to it, every other one to the default char; every
|
||||
// WCTABLE entry encodes to it with best fit, and without best fit only when it round-trips.
|
||||
static int against_source(const std::string& dir)
|
||||
{
|
||||
const unsigned cps[] = {874, 932, 936, 949, 950, 1250, 1251, 1252, 1253, 1254, 1255, 1256, 1257, 1258};
|
||||
int loaded = 0;
|
||||
for (unsigned cp : cps) {
|
||||
SourceTable t;
|
||||
if (!load(dir + "/bestfit" + std::to_string(cp) + ".txt", t)) continue;
|
||||
++loaded;
|
||||
const int before = g_failures;
|
||||
auto is_lead = [&](unsigned b) {
|
||||
for (auto& r : t.leads) if (b >= r.first && b <= r.second) return true;
|
||||
return false;
|
||||
};
|
||||
for (unsigned b = 0; b < 256; ++b) {
|
||||
CHECK(mt_codepage::is_lead_byte(cp, (unsigned char)b) == is_lead(b));
|
||||
if (is_lead(b)) {
|
||||
for (unsigned trail = 1; trail < 256; ++trail) {
|
||||
const unsigned char pair[2] = {(unsigned char)b, (unsigned char)trail};
|
||||
std::u32string out;
|
||||
const bool ok = mt_codepage::decode(cp, pair, 2, out);
|
||||
auto it = t.decode.find(b << 8 | trail);
|
||||
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
|
||||
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
|
||||
std::fprintf(stderr, "cp%u %02x%02x -> %x, want %x\n", cp, b, trail,
|
||||
out.empty() ? 0 : unsigned(out[0]), want);
|
||||
++g_failures;
|
||||
}
|
||||
}
|
||||
} else if (b != 0) {
|
||||
const unsigned char one[1] = {(unsigned char)b};
|
||||
std::u32string out;
|
||||
const bool ok = mt_codepage::decode(cp, one, 1, out);
|
||||
auto it = t.decode.find(b);
|
||||
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
|
||||
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
|
||||
std::fprintf(stderr, "cp%u %02x -> %x, want %x\n", cp, b, out.empty() ? 0 : unsigned(out[0]), want);
|
||||
++g_failures;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (unsigned wc = 1; wc < 0x10000; ++wc) {
|
||||
if (wc >= 0xd800 && wc <= 0xdfff) continue;
|
||||
auto it = t.wctable.find(wc);
|
||||
bool used = false;
|
||||
const std::string fit = enc(cp, std::u32string(1, char32_t(wc)), true, &used);
|
||||
const std::string want_fit = it == t.wctable.end() ? "?" : bytes_of(it->second);
|
||||
const bool exact = it != t.wctable.end() && t.decode.count(it->second) && t.decode[it->second] == wc;
|
||||
const std::string strict = enc(cp, std::u32string(1, char32_t(wc)), false);
|
||||
const std::string want_strict = exact ? bytes_of(it->second) : "?";
|
||||
if (fit != want_fit || used != (it == t.wctable.end()) || strict != want_strict) {
|
||||
std::fprintf(stderr, "cp%u U+%04X -> best fit %zu bytes / strict %zu bytes, mismatch\n", cp, wc,
|
||||
fit.size(), strict.size());
|
||||
++g_failures;
|
||||
}
|
||||
}
|
||||
std::printf("cp%u: %zu byte sequences, %zu WCTABLE entries: %s\n", cp, t.decode.size(), t.wctable.size(),
|
||||
g_failures == before ? "ok" : "FAIL");
|
||||
}
|
||||
return loaded;
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
spot_checks();
|
||||
const char* strict = std::getenv("MT_ASSETS_STRICT");
|
||||
const bool is_strict = strict && std::string(strict) == "1";
|
||||
const int loaded = argc > 1 ? against_source(argv[1]) : 0;
|
||||
if (g_failures) {
|
||||
std::fprintf(stderr, "%d failure(s)\n", g_failures);
|
||||
return 1;
|
||||
}
|
||||
if (loaded < 14) {
|
||||
std::printf("bestfit sources: %d of 14 found (run script/gen_codepage_tables.py)\n", loaded);
|
||||
return is_strict ? 1 : 77;
|
||||
}
|
||||
std::printf("ok\n");
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,227 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate extension/src/codepage/codepage_tables.inc from Microsoft's WindowsBestFit tables.
|
||||
|
||||
The port emulates MultiByteToWideChar / WideCharToMultiByte off Windows. Instead of the host iconv
|
||||
(Android's bionic has no GBK, and iconv tables differ from Windows in the corners), the conversions
|
||||
use Microsoft's own code page definitions, published by unicode.org:
|
||||
|
||||
https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit<cp>.txt
|
||||
|
||||
Each file has an MBTABLE (single bytes), DBCSRANGE/DBCSTABLE (lead byte + trail byte) and a WCTABLE
|
||||
(Unicode -> bytes, including the many-to-one "best fit" entries WideCharToMultiByte uses without
|
||||
WC_NO_BEST_FIT_CHARS). The generated data keeps the byte -> Unicode side and only the WCTABLE entries
|
||||
that differ from its first-match inverse, so the runtime can rebuild the encoder.
|
||||
|
||||
Usage: python3 script/gen_codepage_tables.py [--cache DIR] [--output FILE]
|
||||
Downloads missing files into the cache (build/bestfit by default) and checks their SHA-256.
|
||||
"""
|
||||
import argparse
|
||||
import hashlib
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
URL = "https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit{}.txt"
|
||||
|
||||
# Every code page 40250 EterLib/Util.cpp names (1252 included; CP_ACP maps to it).
|
||||
SHA256 = {
|
||||
874: "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd",
|
||||
932: "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331",
|
||||
936: "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae",
|
||||
949: "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a",
|
||||
950: "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a",
|
||||
1250: "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b",
|
||||
1251: "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7",
|
||||
1252: "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9",
|
||||
1253: "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e",
|
||||
1254: "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a",
|
||||
1255: "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a",
|
||||
1256: "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007",
|
||||
1257: "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366",
|
||||
1258: "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1",
|
||||
}
|
||||
|
||||
NONE = 0xFFFF
|
||||
PAIR = re.compile(rb"^\s*0x([0-9a-fA-F]+)\s+0x([0-9a-fA-F]+)")
|
||||
|
||||
|
||||
def fetch(cp: int, cache: Path) -> bytes:
|
||||
path = cache / f"bestfit{cp}.txt"
|
||||
if not path.is_file():
|
||||
cache.mkdir(parents=True, exist_ok=True)
|
||||
with urllib.request.urlopen(URL.format(cp)) as response:
|
||||
path.write_bytes(response.read())
|
||||
data = path.read_bytes()
|
||||
digest = hashlib.sha256(data).hexdigest()
|
||||
if digest != SHA256[cp]:
|
||||
raise SystemExit(f"{path}: SHA-256 {digest} != {SHA256[cp]}")
|
||||
return data
|
||||
|
||||
|
||||
def parse(cp: int, data: bytes):
|
||||
single = [NONE] * 256
|
||||
leads = [] # [(lo, hi)]
|
||||
dbcs = {} # lead -> {trail: wc}
|
||||
wctable = {} # wc -> mb
|
||||
mb_default = None
|
||||
section = None
|
||||
lead = None
|
||||
for raw in data.splitlines():
|
||||
line = raw.split(b";", 1)[0].strip()
|
||||
if not line:
|
||||
continue
|
||||
head = line.split()[0]
|
||||
if head == b"CODEPAGE":
|
||||
if int(line.split()[1]) != cp:
|
||||
raise SystemExit(f"bestfit{cp}.txt declares {line!r}")
|
||||
continue
|
||||
if head == b"CPINFO":
|
||||
fields = line.split()
|
||||
if int(fields[2], 16) != 0x3F:
|
||||
raise SystemExit(f"cp{cp}: byte default char {fields[2]!r} is not '?'")
|
||||
mb_default = int(fields[3], 16)
|
||||
continue
|
||||
if head in (b"MBTABLE", b"WCTABLE", b"DBCSRANGE"):
|
||||
section = head
|
||||
continue
|
||||
if head == b"DBCSTABLE":
|
||||
section = head
|
||||
match = re.search(rb"LeadByte\s*=\s*0x([0-9a-fA-F]+)", raw)
|
||||
lead = int(match.group(1), 16)
|
||||
dbcs.setdefault(lead, {})
|
||||
continue
|
||||
if head == b"ENDCODEPAGE":
|
||||
break
|
||||
match = PAIR.match(line)
|
||||
if not match:
|
||||
raise SystemExit(f"cp{cp}: unexpected line {raw!r}")
|
||||
a, b = int(match.group(1), 16), int(match.group(2), 16)
|
||||
# cp932 lists its second lead range (0xe0-0xfc) after the first range's DBCSTABLEs.
|
||||
if b"Lead Byte Range" in raw and section in (b"DBCSRANGE", b"DBCSTABLE"):
|
||||
leads.append((a, b))
|
||||
continue
|
||||
if section == b"MBTABLE":
|
||||
single[a] = b
|
||||
elif section == b"DBCSRANGE":
|
||||
leads.append((a, b))
|
||||
elif section == b"DBCSTABLE":
|
||||
dbcs[lead][a] = b
|
||||
elif section == b"WCTABLE":
|
||||
wctable[a] = b
|
||||
if mb_default is None:
|
||||
raise SystemExit(f"cp{cp}: no CPINFO")
|
||||
for lead in dbcs:
|
||||
if not any(lo <= lead <= hi for lo, hi in leads):
|
||||
raise SystemExit(f"cp{cp}: DBCSTABLE 0x{lead:02x} outside the lead ranges {leads}")
|
||||
return single, leads, dbcs, wctable, mb_default
|
||||
|
||||
|
||||
def decode_map(single, dbcs):
|
||||
"""bytes -> wc, bytes as int (single byte < 0x100, pair = lead << 8 | trail), ascending."""
|
||||
out = {}
|
||||
for b, wc in enumerate(single):
|
||||
if wc != NONE:
|
||||
out[b] = wc
|
||||
for lead in sorted(dbcs):
|
||||
for trail in sorted(dbcs[lead]):
|
||||
out[lead << 8 | trail] = dbcs[lead][trail]
|
||||
return out
|
||||
|
||||
|
||||
def default_inverse(decoded):
|
||||
"""The encoder the runtime builds: the first byte sequence (ascending) that decodes to wc."""
|
||||
inverse = {}
|
||||
for mb in sorted(decoded):
|
||||
inverse.setdefault(decoded[mb], mb)
|
||||
return inverse
|
||||
|
||||
|
||||
def encode_overrides(decoded, wctable):
|
||||
"""WCTABLE entries the default inverse gets wrong: 0 round-trip, 1 best fit, 2 not encodable."""
|
||||
inverse = default_inverse(decoded)
|
||||
out = []
|
||||
for wc in sorted(set(inverse) | set(wctable)):
|
||||
want = wctable.get(wc)
|
||||
if want is None:
|
||||
out.append((wc, 0, 2))
|
||||
elif decoded.get(want) != wc:
|
||||
out.append((wc, want, 1))
|
||||
elif inverse.get(wc) != want:
|
||||
out.append((wc, want, 0))
|
||||
return out
|
||||
|
||||
|
||||
def rows(values, per_line=16):
|
||||
return "\n".join("\t" + ", ".join(values[i:i + per_line]) + "," for i in range(0, len(values), per_line))
|
||||
|
||||
|
||||
def emit(cp, single, leads, dbcs, wctable, mb_default):
|
||||
decoded = decode_map(single, dbcs)
|
||||
overrides = encode_overrides(decoded, wctable)
|
||||
# Sanity: the rebuilt encoder must reproduce WCTABLE exactly.
|
||||
inverse = default_inverse(decoded)
|
||||
for wc, mb, kind in overrides:
|
||||
if kind == 2:
|
||||
inverse.pop(wc, None)
|
||||
else:
|
||||
inverse[wc] = mb
|
||||
if inverse != wctable:
|
||||
raise SystemExit(f"cp{cp}: encoder rebuild does not match WCTABLE")
|
||||
|
||||
name = f"kCp{cp}"
|
||||
text = [f"// Code page {cp}: {len(decoded)} byte sequences, {len(wctable)} WCTABLE entries."]
|
||||
text.append(f"const uint16_t {name}Single[256] = {{\n{rows([f'0x{v:04X}' for v in single])}\n}};")
|
||||
lead_ranges = "nullptr"
|
||||
dbcs_rows = "nullptr"
|
||||
dbcs_data = "nullptr"
|
||||
if leads:
|
||||
lead_ranges = f"{name}Leads"
|
||||
text.append(f"const uint8_t {name}Leads[] = {{" + ", ".join(f"0x{lo:02X}, 0x{hi:02X}" for lo, hi in leads) + "};")
|
||||
row_defs = []
|
||||
data = []
|
||||
for lead in sorted(dbcs):
|
||||
trails = dbcs[lead]
|
||||
lo, hi = min(trails), max(trails)
|
||||
row_defs.append(f"\t{{0x{lead:02X}, 0x{lo:02X}, 0x{hi:02X}, {len(data)}}},")
|
||||
data.extend(trails.get(t, NONE) for t in range(lo, hi + 1))
|
||||
dbcs_rows = f"{name}Rows"
|
||||
dbcs_data = f"{name}Dbcs"
|
||||
text.append(f"const DbcsRow {name}Rows[] = {{\n" + "\n".join(row_defs) + "\n};")
|
||||
text.append(f"const uint16_t {name}Dbcs[] = {{\n{rows([f'0x{v:04X}' for v in data])}\n}};")
|
||||
text.append(f"const EncodeOverride {name}Overrides[] = {{\n"
|
||||
+ rows([f"{{0x{wc:04X}, 0x{mb:04X}, {kind}}}" for wc, mb, kind in overrides], 6)
|
||||
+ "\n};" if overrides else f"const EncodeOverride {name}Overrides[] = {{{{0, 0, 2}}}};")
|
||||
count = f"sizeof({name}Overrides) / sizeof({name}Overrides[0])" if overrides else "0"
|
||||
entry = (f"\t{{{cp}, 0x{mb_default:04X}, {name}Single, {lead_ranges}, "
|
||||
f"{len(leads)}, {dbcs_rows}, {len(dbcs) if leads else 0}, {dbcs_data}, "
|
||||
f"{name}Overrides, {count}}},")
|
||||
return "\n".join(text) + "\n", entry, len(overrides)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
parser.add_argument("--cache", type=Path, default=REPO / "build/bestfit")
|
||||
parser.add_argument("--output", type=Path, default=REPO / "extension/src/codepage/codepage_tables.inc")
|
||||
args = parser.parse_args()
|
||||
|
||||
body = []
|
||||
entries = []
|
||||
for cp in sorted(SHA256):
|
||||
parsed = parse(cp, fetch(cp, args.cache))
|
||||
text, entry, n_over = emit(cp, *parsed)
|
||||
body.append(text)
|
||||
entries.append(entry)
|
||||
print(f"cp{cp}: {n_over} encode overrides", file=sys.stderr)
|
||||
header = ("// Generated by script/gen_codepage_tables.py from Microsoft's WindowsBestFit tables\n"
|
||||
"// (unicode.org MAPPINGS/VENDORS/MICSFT/WindowsBestFit). Do not edit.\n"
|
||||
"// 0xFFFF marks a byte / pair with no mapping. Included by codepage.cpp only.\n")
|
||||
table = "const CodePageTable kTables[] = {\n" + "\n".join(entries) + "\n};\n"
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(header + "\n" + "\n".join(body) + "\n" + table, encoding="ascii")
|
||||
print(f"wrote {args.output}", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user