codepage: replace iconv with bundled WindowsBestFit tables

mt_codepage implements code pages 874/932/936/949/950/1250-1258 from
Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py,
SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/
IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build
no longer depends on iconv. port.codepage checks every table against
its source file.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-28 18:45:49 +09:00
co-authored by Claude Opus 5.5
parent 07c101fdc8
commit 0b701d88c3
12 changed files with 7068 additions and 191 deletions
+11 -2
View File
@@ -353,7 +353,16 @@ extension/third_party/cpython-2.7.18/ # 静态库(2P 批次
| ~~Linux x86_64~~ | 不做(2026-09-23 定,见第 1 节) |
| ~~Windows x64~~ | 不做(2026-09-23 定,见第 1 节)。记录原因备查:上游 2.7 只支持 MSVC + `PC/pyconfig.h`,而本仓库的 Windows 门禁是 mingw-w64 交叉编译,`PC/pyconfig.h` 的 gnu-win32 分支不设 `HAVE_UNISTD_H`,`posixmodule.c`/`dynload_win.c` 与 MinGW 头文件冲突。因此 mingw 门禁下不建 `mtpython`,`port_logic`/`port_platform` 自动排除 ScriptLib |
### 已知问题:Android 的 GBK 编解码(挂起,2026-09-23)
### 已解决:Android 的 GBK 编解码(2026-09-28)
`extension/src/codepage`(`mt_codepage`)按微软 WindowsBestFit 表实现 874/932/936/949/950/1250–1258,
`Win32Crt.cpp` 的 `MultiByteToWideChar`/`WideCharToMultiByte`/`IsDBCSLeadByteEx` 和 `net/text_codec.cpp` 都改用它,
iconv 已全部移除。表由 `script/gen_codepage_tables.py` 从 unicode.org 下载(SHA-256 固定)生成;
`port.codepage` 测试逐条校验 14 张表与源文件一致(解码全部单字节/双字节序列,编码全部 BMP 码位,含/不含 best fit)。
与 macOS iconv 一次性逐字对比:CP936 22,111 个码位一致,差异只在 Windows 特有的部分——0x80→€、EUDC 用户造字区映射到
私用区(U+E000–F8FF,1,955 个)、A2E3/A8BF 两处;1250/1251/1253/1254 只差未定义字节(Windows 映射为 C1 控制码)。下面是当时的分析记录。
#### 原记录(2026-09-23)
**优先级:macOS 先跑通,Android 后面再说**(用户 2026-09-23 定)。
@@ -515,7 +524,7 @@ extension/third_party/cpython-2.7.18/ # 静态库(2P 批次
`Eternexus/root/*.msm` 的路径、大小、sha256,结果在 `audit/assets/40250-client.json`(223 个文件,1,407,474,045
字节)。文件缺失、内容变化或清单外多出的文件都会让 verify 失败。
已知问题(不是本批引入的):完整 Android 链接在 API 24 上失败,原因是 `extension/src/net/text_codec.cpp` 用了 iconv;
已知问题(不是本批引入的,2026-09-28 已随 `mt_codepage` 解决):完整 Android 链接在 API 24 上失败,原因是 `extension/src/net/text_codec.cpp` 用了 iconv;
本批的 mtproto、`pack40250_node`、`asset_io`、`proto_node` 都能在 Android 上编译。
## 6. 进度与 port-map 基线
+3 -7
View File
@@ -14,6 +14,8 @@ add_subdirectory(godot-cpp)
# aliases mt3p::sodium / mt3p::zstd / mt3p::minilzo. See docs/THIRD-PARTY.md.
add_subdirectory(third_party)
add_subdirectory(src/codepage)
# --- mtnet: Metin2 net protocol core (no godot-cpp dep; libsodium only) ---
add_library(mtnet STATIC
src/net/text_codec.cpp
@@ -27,13 +29,7 @@ add_library(mtnet STATIC
src/net/classic/classic_cipher.cpp
)
target_include_directories(mtnet PUBLIC src/net)
target_link_libraries(mtnet PUBLIC mt3p::sodium mt3p::minilzo mt3p::cryptopp)
# macOS ships iconv as a system library rather than libc. Linux/glibc and
# Windows use the platform implementation without this extra link item.
find_library(MT_ICONV_LIBRARY NAMES iconv)
if(MT_ICONV_LIBRARY)
target_link_libraries(mtnet PUBLIC ${MT_ICONV_LIBRARY})
endif()
target_link_libraries(mtnet PUBLIC mt3p::sodium mt3p::minilzo mt3p::cryptopp mt_codepage)
target_compile_features(mtnet PUBLIC cxx_std_20)
# §1.5: non-EUROPE CG_CLIENT_VERSION timestamp, in C __TIMESTAMP__ form
# ("Www Mmm dd hh:mm:ss yyyy"); regenerated on each CMake configure.
+8
View File
@@ -0,0 +1,8 @@
# mt_codepage — Windows ANSI code pages (874/932/936/949/950/1250-1258) from Microsoft's WindowsBestFit
# tables; codepage_tables.inc is generated by script/gen_codepage_tables.py. Used by port_logic
# (Win32Crt MultiByteToWideChar / WideCharToMultiByte) and mtnet (40250 CP936 wire text), in place of
# the host iconv.
add_library(mt_codepage STATIC codepage.cpp)
target_include_directories(mt_codepage PUBLIC ${CMAKE_CURRENT_SOURCE_DIR})
target_compile_features(mt_codepage PUBLIC cxx_std_17)
set_target_properties(mt_codepage PROPERTIES POSITION_INDEPENDENT_CODE ON)
+169
View File
@@ -0,0 +1,169 @@
#include "codepage.h"
#include <algorithm>
#include <cstdint>
#include <memory>
#include <mutex>
namespace mt_codepage {
namespace {
constexpr uint16_t kNone = 0xFFFF;
struct DbcsRow {
uint8_t lead;
uint8_t trail_lo;
uint8_t trail_hi;
uint32_t offset; // into the Dbcs array, trail_lo first
};
// A WCTABLE entry the first-match inverse of the byte tables gets wrong.
struct EncodeOverride {
uint16_t wc;
uint16_t mb;
uint8_t kind; // 0 exact (round-trips), 1 best fit only, 2 not encodable
};
struct CodePageTable {
unsigned cp;
uint16_t mb_default; // Unicode char for an unmapped byte / pair (CPINFO)
const uint16_t* single;
const uint8_t* leads; // lo, hi pairs
size_t lead_count;
const DbcsRow* rows;
size_t row_count;
const uint16_t* dbcs;
const EncodeOverride* overrides;
size_t override_count;
};
#include "codepage_tables.inc"
constexpr size_t kTableCount = sizeof(kTables) / sizeof(kTables[0]);
const CodePageTable* find(unsigned cp)
{
for (const CodePageTable& t : kTables)
if (t.cp == cp) return &t;
return nullptr;
}
bool lead(const CodePageTable& t, unsigned char b)
{
for (size_t i = 0; i < t.lead_count; ++i)
if (b >= t.leads[i * 2] && b <= t.leads[i * 2 + 1]) return true;
return false;
}
uint16_t pair(const CodePageTable& t, unsigned char l, unsigned char trail)
{
for (size_t i = 0; i < t.row_count; ++i) {
const DbcsRow& r = t.rows[i];
if (r.lead != l) continue;
if (trail < r.trail_lo || trail > r.trail_hi) return kNone;
return t.dbcs[r.offset + (trail - r.trail_lo)];
}
return kNone;
}
// Exact encoder (BMP code point -> bytes, kNone when unmapped), built on first use per code page.
struct Encoder {
std::once_flag once;
std::unique_ptr<uint16_t[]> exact;
};
Encoder g_encoders[kTableCount];
const uint16_t* exact_encoder(const CodePageTable& t)
{
Encoder& e = g_encoders[&t - kTables];
std::call_once(e.once, [&] {
e.exact.reset(new uint16_t[0x10000]);
std::fill_n(e.exact.get(), 0x10000, kNone);
uint16_t* map = e.exact.get();
// First match in ascending byte order, as the generator assumed.
for (unsigned b = 0; b < 256; ++b)
if (t.single[b] != kNone && map[t.single[b]] == kNone) map[t.single[b]] = uint16_t(b);
for (size_t i = 0; i < t.row_count; ++i) {
const DbcsRow& r = t.rows[i];
for (unsigned trail = r.trail_lo; trail <= r.trail_hi; ++trail) {
const uint16_t wc = t.dbcs[r.offset + (trail - r.trail_lo)];
if (wc != kNone && map[wc] == kNone) map[wc] = uint16_t(r.lead << 8 | trail);
}
}
for (size_t i = 0; i < t.override_count; ++i) {
const EncodeOverride& o = t.overrides[i];
if (o.kind == 0) map[o.wc] = o.mb;
else if (o.kind == 2) map[o.wc] = kNone;
}
});
return e.exact.get();
}
uint16_t best_fit(const CodePageTable& t, char32_t c)
{
const EncodeOverride* end = t.overrides + t.override_count;
const EncodeOverride* it = std::lower_bound(t.overrides, end, c,
[](const EncodeOverride& o, char32_t v) { return o.wc < v; });
return it != end && it->wc == c && it->kind == 1 ? it->mb : kNone;
}
} // namespace
bool supported(unsigned cp)
{
return find(cp) != nullptr;
}
bool is_lead_byte(unsigned cp, unsigned char b)
{
const CodePageTable* t = find(cp);
return t && lead(*t, b);
}
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out)
{
const CodePageTable* t = find(cp);
if (!t) return false;
bool ok = true;
for (size_t i = 0; i < n;) {
const unsigned char b = p[i];
uint16_t wc = kNone;
if (lead(*t, b) && i + 1 < n && p[i + 1] != 0) {
wc = pair(*t, b, p[i + 1]);
i += 2;
} else {
wc = t->single[b];
++i;
}
if (wc == kNone) {
wc = t->mb_default;
ok = false;
}
out.push_back(wc);
}
return ok;
}
void encode(unsigned cp, std::u32string_view in, std::string& out, bool allow_best_fit, bool* used_default)
{
const CodePageTable* t = find(cp);
const uint16_t* exact = t ? exact_encoder(*t) : nullptr;
for (char32_t c : in) {
uint16_t mb = kNone;
if (exact && c < 0x10000) {
mb = exact[c];
if (mb == kNone && allow_best_fit) mb = best_fit(*t, c);
}
if (mb == kNone) {
out.push_back('?');
if (used_default) *used_default = true;
} else if (mb > 0xFF) {
out.push_back(char(mb >> 8));
out.push_back(char(mb & 0xFF));
} else {
out.push_back(char(mb));
}
}
}
} // namespace mt_codepage
+29
View File
@@ -0,0 +1,29 @@
#pragma once
// Windows ANSI code pages without the host iconv: MultiByteToWideChar / WideCharToMultiByte semantics
// driven by Microsoft's WindowsBestFit tables (script/gen_codepage_tables.py). One implementation for
// macOS, Linux, Android and iOS, so every platform decodes the 40250 locale and wire text the same way.
//
// Covered: 874, 932, 936, 949, 950, 1250-1258 (every code page 40250 EterLib/Util.cpp names).
// UTF-8 is not handled here.
#include <cstddef>
#include <string>
#include <string_view>
namespace mt_codepage {
bool supported(unsigned cp);
// Byte string -> code points, as MultiByteToWideChar. A lead byte takes the next non-NUL byte as its
// trail; an unmapped byte or pair becomes the code page's default char (U+30FB for 932, else '?').
// Returns false when a default char was substituted (MB_ERR_INVALID_CHARS would fail).
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out);
// Code points -> bytes, as WideCharToMultiByte. With best_fit (no WC_NO_BEST_FIT_CHARS) a character
// without an exact mapping may take its WCTABLE best fit ("Ā" -> "A" on 1252); otherwise, and for
// anything unmapped, '?' is written and *used_default set.
void encode(unsigned cp, std::u32string_view in, std::string& out, bool best_fit, bool* used_default);
bool is_lead_byte(unsigned cp, unsigned char b);
} // namespace mt_codepage
File diff suppressed because it is too large Load Diff
+52 -57
View File
@@ -1,16 +1,12 @@
#include "text_codec.h"
#include <algorithm>
#include <cerrno>
#include <cstring>
#if defined(_WIN32)
#include <windows.h>
#elif defined(__has_include)
#if __has_include(<iconv.h>)
#include <iconv.h>
#define MT_TEXT_CODEC_HAS_ICONV 1
#endif
#else
// CP936 from Microsoft's own table, not the host iconv (bionic has no GBK).
#include "codepage.h"
#endif
namespace mtnet {
@@ -69,61 +65,66 @@ bool convert_from_windows(std::string_view input, std::string &out) {
return WideCharToMultiByte(CP_UTF8, 0, wide.data(), wide_len, out.data(), utf8_len,
nullptr, nullptr) == utf8_len;
}
#elif defined(MT_TEXT_CODEC_HAS_ICONV)
bool convert_iconv(std::string_view input, const char *from, const char *to, std::string &out) {
iconv_t cd = iconv_open(to, from);
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
std::string buffer(std::max<size_t>(input.size() * 4 + 16, 32), '\0');
char *in_ptr = const_cast<char *>(input.data());
size_t in_left = input.size();
char *out_ptr = buffer.data();
size_t out_left = buffer.size();
while (in_left != 0) {
const size_t rc = iconv(cd, &in_ptr, &in_left, &out_ptr, &out_left);
if (rc == static_cast<size_t>(-1)) {
iconv_close(cd);
#else
// Mirrors the Windows branch: CP936 without best-fit substitutions, and a strict decode.
constexpr unsigned kWireCodePage = 936;
bool decode_utf8_strict(std::string_view input, std::u32string &out) {
for (size_t i = 0; i < input.size();) {
const size_t seq = utf8_sequence_length(input.data() + i, input.size() - i);
if (seq == 0) return false;
const unsigned char c = static_cast<unsigned char>(input[i]);
char32_t cp = seq == 1 ? c : seq == 2 ? (c & 0x1F) : seq == 3 ? (c & 0x0F) : (c & 0x07);
for (size_t k = 1; k < seq; ++k) {
const unsigned char t = static_cast<unsigned char>(input[i + k]);
if ((t & 0xC0) != 0x80) return false;
cp = (cp << 6) | (t & 0x3F);
}
if ((seq == 3 && cp < 0x800) || (seq == 4 && (cp < 0x10000 || cp > 0x10FFFF)) ||
(cp >= 0xD800 && cp <= 0xDFFF)) {
return false;
}
out.push_back(cp);
i += seq;
}
// Flush stateful encodings, even though GB2312 itself is stateless.
iconv(cd, nullptr, nullptr, &out_ptr, &out_left);
iconv_close(cd);
out.assign(buffer.data(), static_cast<size_t>(out_ptr - buffer.data()));
return true;
}
void append_utf8(char32_t cp, std::string &out) {
if (cp < 0x80) {
out.push_back(static_cast<char>(cp));
} else if (cp < 0x800) {
out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
} else if (cp < 0x10000) {
out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
} else {
out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
}
}
bool encode_one(std::string_view input, std::string &out) {
return convert_iconv(input, "UTF-8", "GB2312", out) ||
convert_iconv(input, "UTF-8", "GBK", out);
std::u32string wide;
if (!decode_utf8_strict(input, wide)) return false;
bool used_default = false;
out.clear();
mt_codepage::encode(kWireCodePage, wide, out, false, &used_default);
return !used_default;
}
bool decode_wire(std::string_view input, std::string &out) {
return convert_iconv(input, "GB2312", "UTF-8", out) ||
convert_iconv(input, "GBK", "UTF-8", out);
}
#else
bool encode_one(std::string_view input, std::string &out) {
// A platform without either Windows code-page APIs or iconv must never
// send UTF-8 bytes into a 40250 fixed legacy field. ASCII is lossless;
// non-ASCII is reported as unrepresentable so wire_text_fits() rejects it
// and to_wire() can safely substitute '?' rather than leaking a 3-byte
// UTF-8 sequence into a 24-byte name field.
const size_t n = utf8_byte_truncate(input.data(), input.size(), input.size());
if (n != input.size()) return false;
for (unsigned char c : input) {
if (c >= 0x80) return false;
std::u32string wide;
if (!mt_codepage::decode(kWireCodePage, reinterpret_cast<const unsigned char *>(input.data()),
input.size(), wide)) {
return false;
}
out.assign(input.data(), input.size());
return true;
}
bool decode_wire(std::string_view input, std::string &out) {
// The fallback decoder is deliberately conservative for the same reason:
// returning raw high-bit bytes would create invalid UTF-8 in the client.
for (unsigned char c : input) {
if (c >= 0x80) return false;
}
out.assign(input.data(), input.size());
out.clear();
for (char32_t c : wide) append_utf8(c, out);
return true;
}
#endif
@@ -176,7 +177,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
size_t n = 0;
while (n < cap && bytes[n] != '\0') ++n;
std::string decoded;
#if defined(_WIN32) || defined(MT_TEXT_CODEC_HAS_ICONV)
// A fixed field can arrive without a terminator and a caller may provide a
// larger view than the actual field. If that view ends on a legacy lead byte,
// drop only the malformed suffix and preserve the complete preceding text.
@@ -188,11 +188,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
#endif
if (ok) return decoded;
}
#else
if (decode_wire(std::string_view(bytes, n), decoded)) {
return decoded;
}
#endif
// Keep malformed input visible without leaking legacy bytes as invalid UTF-8.
decoded.clear();
for (size_t i = 0; i < n; ++i) {
+13 -5
View File
@@ -31,12 +31,12 @@ target_include_directories(port_logic PUBLIC ${MT_PORT_SHIMS})
# Libraries 40250 links that are vendored as-is: <lzo/lzo1x.h> (EterBase/lzo.h), <cryptopp/*>
# (EterBase/cipher.h), <Python-2.7/*> (ScriptLib).
target_link_libraries(port_logic PUBLIC mt3p::minilzo mt3p::cryptopp)
# common/Win32Crt.cpp converts non-UTF-8 code pages with iconv. macOS ships it as a separate library;
# extension/CMakeLists.txt finds it for mtnet, but the port-only build (script/port_gate.sh) never reads that file.
find_library(MT_ICONV_LIBRARY NAMES iconv)
if(MT_ICONV_LIBRARY)
target_link_libraries(port_logic PUBLIC ${MT_ICONV_LIBRARY})
# common/Win32Crt.cpp converts ANSI code pages with mt_codepage. extension/CMakeLists.txt adds it for
# mtnet; the port-only build (script/port_gate.sh, native_render) never reads that file.
if(NOT TARGET mt_codepage)
add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/../codepage ${CMAKE_CURRENT_BINARY_DIR}/../codepage)
endif()
target_link_libraries(port_logic PUBLIC mt_codepage)
if(TARGET mtpython)
target_link_libraries(port_logic PUBLIC mt3p::python)
endif()
@@ -173,6 +173,14 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})
set_tests_properties(port.eterpack PROPERTIES SKIP_RETURN_CODE 77)
# mt_codepage against Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py caches them
# in build/bestfit); skipped (77) without them unless MT_ASSETS_STRICT=1.
add_executable(port_codepage_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/codepage_test.cpp)
target_link_libraries(port_codepage_test PRIVATE mt_codepage)
add_test(NAME port.codepage COMMAND $<TARGET_FILE:port_codepage_test>
${CMAKE_CURRENT_SOURCE_DIR}/../../../build/bestfit)
set_tests_properties(port.codepage PROPERTIES SKIP_RETURN_CODE 77)
# 批次 2V0-g: the 40250 font path over the FreeType GDI emulation; needs a system Tahoma.
add_executable(port_text_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/port_text_test.cpp)
target_link_libraries(port_text_test PRIVATE port_platform)
+37 -118
View File
@@ -11,11 +11,11 @@
#include <thread>
#include <vector>
#if __has_include(<iconv.h>)
#include <iconv.h>
#define MT_WIN32CRT_HAS_ICONV 1
#endif
#include <glob.h>
// ANSI code pages come from Microsoft's own tables, not the host iconv (bionic has no GBK).
#include "codepage.h"
#include <algorithm>
#include <dirent.h>
#include <fnmatch.h>
#include <sys/stat.h>
#include <unistd.h>
@@ -103,17 +103,30 @@ HANDLE FindFirstFile(const char* pattern, WIN32_FIND_DATA* data)
std::string normalized(pattern);
for (char& ch : normalized)
if (ch == '\\') ch = '/';
glob_t matches{};
const int result = glob(normalized.c_str(), 0, nullptr, &matches);
if (result != 0 || matches.gl_pathc == 0)
// Win32 wildcards only appear in the last component; match it against the directory like
// glob(3) would (sorted, '*' skips dot files). glob itself is API 28+ on Android.
std::vector<std::string> paths;
const size_t slash = normalized.rfind('/');
const std::string dir = slash == std::string::npos ? std::string() : normalized.substr(0, slash + 1);
const std::string name = normalized.substr(dir.size());
if (name.find_first_of("*?[") == std::string::npos)
{
globfree(&matches);
return INVALID_HANDLE_VALUE;
struct stat st;
if (::stat(normalized.c_str(), &st) == 0)
paths.push_back(normalized);
}
else if (DIR* d = ::opendir(dir.empty() ? "." : dir.c_str()))
{
while (const dirent* entry = ::readdir(d))
if (::fnmatch(name.c_str(), entry->d_name, FNM_PERIOD) == 0)
paths.push_back(dir + entry->d_name);
::closedir(d);
std::sort(paths.begin(), paths.end());
}
if (paths.empty())
return INVALID_HANDLE_VALUE;
auto* state = new FindState;
for (size_t i = 0; i < matches.gl_pathc; ++i)
state->paths.emplace_back(matches.gl_pathv[i]);
globfree(&matches);
state->paths = std::move(paths);
fill_find_data(state->paths[0], data);
state->next = 1;
return state;
@@ -199,13 +212,8 @@ DWORD GetTickCount()
namespace {
// Windows-1252 0x80..0x9F; the other bytes map to the same code point (0 marks the undefined ones).
const char32_t kCp1252High[32] = {
0x20AC, 0, 0x201A, 0x0192, 0x201E, 0x2026, 0x2020, 0x2021, 0x02C6, 0x2030, 0x0160, 0x2039, 0x0152, 0, 0x017D, 0,
0, 0x2018, 0x2019, 0x201C, 0x201D, 0x2022, 0x2013, 0x2014, 0x02DC, 0x2122, 0x0161, 0x203A, 0x0153, 0, 0x017E, 0x0178,
};
bool is_cp1252(UINT cp) { return cp == CP_ACP || cp == 1252; }
// CP_ACP is the Western system code page here, as on the 40250 target machines.
UINT ansi(UINT cp) { return cp == CP_ACP ? 1252 : cp; }
bool decode_utf8(const unsigned char* p, size_t n, std::u32string& out)
{
@@ -237,108 +245,25 @@ void encode_utf8(char32_t cp, std::string& out)
else { out.push_back(char(0xF0 | (cp >> 18))); out.push_back(char(0x80 | ((cp >> 12) & 0x3F))); out.push_back(char(0x80 | ((cp >> 6) & 0x3F))); out.push_back(char(0x80 | (cp & 0x3F))); }
}
#if defined(MT_WIN32CRT_HAS_ICONV)
std::string iconv_name(UINT cp)
{
switch (cp) {
case 949: return "CP949";
case 936: return "GBK";
case 950: return "BIG5";
case 932: return "SHIFT_JIS";
case 874: return "CP874";
default: return "CP" + std::to_string(cp);
}
}
// One character at a time, so an undecodable byte becomes one default char like Windows does.
bool iconv_decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
{
iconv_t cd = iconv_open("UTF-32LE", iconv_name(cp).c_str());
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
for (size_t i = 0; i < n;) {
unsigned char wide[16];
bool done = false;
for (size_t len = 1; len <= 4 && i + len <= n && !done; ++len) {
iconv(cd, nullptr, nullptr, nullptr, nullptr);
char* in = const_cast<char*>(reinterpret_cast<const char*>(p + i));
size_t in_left = len;
char* o = reinterpret_cast<char*>(wide);
size_t o_left = sizeof(wide);
if (iconv(cd, &in, &in_left, &o, &o_left) != size_t(-1) && in_left == 0 && o_left < sizeof(wide)) {
for (size_t k = 0; k + 4 <= sizeof(wide) - o_left; k += 4)
out.push_back(char32_t(wide[k] | (wide[k + 1] << 8) | (wide[k + 2] << 16) | (unsigned(wide[k + 3]) << 24)));
i += len;
done = true;
}
}
if (!done) { out.push_back(0xFFFD); ++i; }
}
iconv_close(cd);
return true;
}
bool iconv_encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
{
iconv_t cd = iconv_open(iconv_name(cp).c_str(), "UTF-32LE");
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
for (char32_t c : in) {
unsigned char src[4] = {static_cast<unsigned char>(c), static_cast<unsigned char>(c >> 8), static_cast<unsigned char>(c >> 16), static_cast<unsigned char>(c >> 24)};
char buf[16];
char* ip = reinterpret_cast<char*>(src);
size_t il = 4;
char* op = buf;
size_t ol = sizeof(buf);
iconv(cd, nullptr, nullptr, nullptr, nullptr);
if (iconv(cd, &ip, &il, &op, &ol) != size_t(-1) && il == 0) {
out.append(buf, sizeof(buf) - ol);
} else {
out.push_back('?');
*used_default = true;
}
}
iconv_close(cd);
return true;
}
#endif
bool decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
{
if (cp == CP_UTF8) return decode_utf8(p, n, out);
if (is_cp1252(cp)) {
for (size_t i = 0; i < n; ++i) {
const unsigned char c = p[i];
out.push_back(c >= 0x80 && c < 0xA0 ? (kCp1252High[c - 0x80] ? kCp1252High[c - 0x80] : char32_t(c)) : char32_t(c));
}
return true;
}
#if defined(MT_WIN32CRT_HAS_ICONV)
if (iconv_decode(cp, p, n, out)) return true;
#endif
// No converter for this code page: ASCII passes, the rest becomes U+FFFD.
if (mt_codepage::supported(ansi(cp))) return mt_codepage::decode(ansi(cp), p, n, out);
// No table for this code page: ASCII passes, the rest becomes U+FFFD.
for (size_t i = 0; i < n; ++i) out.push_back(p[i] < 0x80 ? char32_t(p[i]) : char32_t(0xFFFD));
return true;
}
void encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
void encode(UINT cp, DWORD flags, const std::u32string& in, std::string& out, bool* used_default)
{
if (cp == CP_UTF8) {
for (char32_t c : in) encode_utf8(c, out);
return;
}
if (is_cp1252(cp)) {
for (char32_t c : in) {
if (c < 0x80 || (c >= 0xA0 && c < 0x100)) { out.push_back(char(c)); continue; }
int hit = -1;
for (int k = 0; k < 32; ++k)
if (kCp1252High[k] == c) { hit = k; break; }
if (hit >= 0) out.push_back(char(0x80 + hit));
else { out.push_back('?'); *used_default = true; }
}
if (mt_codepage::supported(ansi(cp))) {
mt_codepage::encode(ansi(cp), in, out, (flags & WC_NO_BEST_FIT_CHARS) == 0, used_default);
return;
}
#if defined(MT_WIN32CRT_HAS_ICONV)
if (iconv_encode(cp, in, out, used_default)) return;
#endif
for (char32_t c : in) {
if (c < 0x80) out.push_back(char(c));
else { out.push_back('?'); *used_default = true; }
@@ -361,7 +286,7 @@ int MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int
return int(wide.size());
}
int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWideChar,
int WideCharToMultiByte(UINT CodePage, DWORD dwFlags, LPCWSTR lpWideCharStr, int cchWideChar,
LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar, LPBOOL lpUsedDefaultChar)
{
if (!lpWideCharStr || cchWideChar == 0 || cbMultiByte < 0) return 0;
@@ -372,14 +297,14 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
for (size_t i = 0; i < n; ++i) wide[i] = char32_t(lpWideCharStr[i]);
std::string bytes;
bool used_default = false;
encode(CodePage, wide, bytes, &used_default);
encode(CodePage, dwFlags, wide, bytes, &used_default);
if (used_default && lpDefaultChar && CodePage != CP_UTF8) {
// '?' was the placeholder; the caller's default char replaces it only where a fallback happened.
std::string redo;
for (char32_t c : wide) {
std::string one;
bool miss = false;
encode(CodePage, std::u32string(1, c), one, &miss);
encode(CodePage, dwFlags, std::u32string(1, c), one, &miss);
redo += miss ? std::string(1, *lpDefaultChar) : one;
}
bytes.swap(redo);
@@ -393,13 +318,7 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
BOOL IsDBCSLeadByteEx(UINT CodePage, BYTE TestChar)
{
switch (CodePage) {
case 932: return (TestChar >= 0x81 && TestChar <= 0x9F) || (TestChar >= 0xE0 && TestChar <= 0xFC);
case 936:
case 949:
case 950: return TestChar >= 0x81 && TestChar <= 0xFE;
default: return FALSE;
}
return mt_codepage::is_lead_byte(ansi(CodePage), TestChar) ? TRUE : FALSE;
}
LPSTR CharNextExA(WORD CodePage, LPCSTR lpCurrentChar, DWORD)
+2 -2
View File
@@ -104,8 +104,8 @@ void DeleteCriticalSection(LPCRITICAL_SECTION cs);
void EnterCriticalSection(LPCRITICAL_SECTION cs);
void LeaveCriticalSection(LPCRITICAL_SECTION cs);
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 and
// CP_ACP/1252 are converted directly; other code pages go through iconv where the platform has it.
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 is converted
// directly; CP_ACP (= 1252) and the other ANSI code pages use Microsoft's tables (codepage/codepage.h).
#define CP_ACP 0
#define CP_UTF8 65001
#define MB_PRECOMPOSED 0x00000001
+214
View File
@@ -0,0 +1,214 @@
// codepage_test — mt_codepage against Microsoft's WindowsBestFit source tables, plus spot checks of the
// characters each shipped 40250 locale needs.
//
// Argument: the directory holding bestfit<cp>.txt (script/gen_codepage_tables.py downloads them to
// build/bestfit). Without it only the spot checks run and the test exits 77 (ctest SKIP), unless
// MT_ASSETS_STRICT=1.
#include "codepage.h"
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <fstream>
#include <map>
#include <sstream>
#include <string>
#include <vector>
static int g_failures = 0;
#define CHECK(cond) \
do { \
if (!(cond)) { \
std::fprintf(stderr, "%s:%d: CHECK(%s)\n", __FILE__, __LINE__, #cond); \
++g_failures; \
} \
} while (0)
static std::u32string dec(unsigned cp, const char* bytes, bool* ok = nullptr)
{
std::u32string out;
const bool r = mt_codepage::decode(cp, reinterpret_cast<const unsigned char*>(bytes), std::strlen(bytes), out);
if (ok) *ok = r;
return out;
}
static std::string enc(unsigned cp, std::u32string in, bool best_fit = false, bool* used_default = nullptr)
{
std::string out;
bool used = false;
mt_codepage::encode(cp, in, out, best_fit, &used);
if (used_default) *used_default = used;
return out;
}
static void spot_checks()
{
// zh (936): "中文" D6D0 CEC4, and the Windows-only single byte 0x80 = euro.
CHECK(dec(936, "\xd6\xd0\xce\xc4") == U"中文");
CHECK(enc(936, U"中文") == "\xd6\xd0\xce\xc4");
CHECK(dec(936, "\x80") == U"€");
// GBK extension beyond GB2312 (U+4E02 = 81 40), both directions.
CHECK(dec(936, "\x81\x40") == U"丂");
CHECK(enc(936, U"丂") == "\x81\x40");
// A lead byte with nothing after it is one default char and reports failure.
bool ok = true;
CHECK(dec(936, "a\xd6", &ok) == U"a?" && !ok);
// An unmapped pair is consumed whole (932 uses U+30FB as its default).
CHECK(dec(932, "\x85\x40") == U"・");
CHECK(dec(932, "\x82\xa0") == U"あ");
// The shipped European locales.
CHECK(dec(1252, "\x80\xe9") == U"€é"); // en/de/fr/...: euro, é
CHECK(dec(1252, "\x81") == U"\u0081"); // undefined byte keeps its value, as Windows
CHECK(dec(1250, "\xb9\xb3\x9c") == U"ąłś"); // pl/cz/hu/ro: ą ł ś
CHECK(dec(1251, "\xc6\xe8") == U"Жи"); // ru: Жи
CHECK(dec(1253, "\xd9\xe1") == U"Ωα"); // gr: Ωα
CHECK(dec(1254, "\xf0\xfe\xdd") == U"ğşİ"); // tr: ğ ş İ
CHECK(enc(1250, U"ąłś") == "\xb9\xb3\x9c");
CHECK(enc(1251, U"Жи") == "\xc6\xe8");
CHECK(enc(1253, U"Ωα") == "\xd9\xe1");
CHECK(enc(1254, U"ğşİ") == "\xf0\xfe\xdd");
// Best fit only when WC_NO_BEST_FIT_CHARS is absent: 1252 has no "Ā", its best fit is "A".
bool used = false;
CHECK(enc(1252, U"Ā", true, &used) == "A" && !used);
CHECK(enc(1252, U"Ā", false, &used) == "?" && used);
// Unmapped either way; beyond the BMP too.
CHECK(enc(1251, U"中", true, &used) == "?" && used);
CHECK(enc(936, U"\U0001F600", true, &used) == "?" && used);
CHECK(mt_codepage::is_lead_byte(936, 0x81) && !mt_codepage::is_lead_byte(936, 0x80));
CHECK(mt_codepage::is_lead_byte(932, 0xe0) && !mt_codepage::is_lead_byte(932, 0xa1));
CHECK(!mt_codepage::is_lead_byte(1250, 0xb9));
CHECK(!mt_codepage::supported(65001) && !mt_codepage::supported(437));
}
struct SourceTable {
unsigned mb_default = 0x3f;
std::map<unsigned, unsigned> decode; // bytes (pair = lead << 8 | trail) -> wc
std::map<unsigned, unsigned> wctable;
std::vector<std::pair<unsigned, unsigned>> leads;
};
static bool load(const std::string& path, SourceTable& t)
{
std::ifstream in(path, std::ios::binary);
if (!in) return false;
std::string raw, section;
unsigned lead = 0;
while (std::getline(in, raw)) {
std::string line = raw.substr(0, raw.find(';'));
std::istringstream words(line);
std::string head;
if (!(words >> head)) continue;
if (head == "CPINFO") {
std::string size, byte_default, wc_default;
words >> size >> byte_default >> wc_default;
t.mb_default = unsigned(std::stoul(wc_default, nullptr, 16));
continue;
}
if (head == "MBTABLE" || head == "WCTABLE" || head == "DBCSRANGE") { section = head; continue; }
if (head == "DBCSTABLE") {
section = head;
lead = unsigned(std::stoul(raw.substr(raw.find("0x", raw.find("LeadByte"))), nullptr, 16));
continue;
}
if (head == "ENDCODEPAGE") break;
if (head.rfind("0x", 0) != 0) continue;
std::string second;
words >> second;
const unsigned a = unsigned(std::stoul(head, nullptr, 16));
const unsigned b = unsigned(std::stoul(second, nullptr, 16));
if (raw.find("Lead Byte Range") != std::string::npos) t.leads.push_back({a, b});
else if (section == "MBTABLE") t.decode[a] = b;
else if (section == "DBCSTABLE") t.decode[lead << 8 | a] = b;
else if (section == "WCTABLE") t.wctable[a] = b;
}
return !t.decode.empty() && !t.wctable.empty();
}
static std::string bytes_of(unsigned mb)
{
return mb > 0xff ? std::string{char(mb >> 8), char(mb & 0xff)} : std::string(1, char(mb));
}
// Every byte / pair the source defines decodes to it, every other one to the default char; every
// WCTABLE entry encodes to it with best fit, and without best fit only when it round-trips.
static int against_source(const std::string& dir)
{
const unsigned cps[] = {874, 932, 936, 949, 950, 1250, 1251, 1252, 1253, 1254, 1255, 1256, 1257, 1258};
int loaded = 0;
for (unsigned cp : cps) {
SourceTable t;
if (!load(dir + "/bestfit" + std::to_string(cp) + ".txt", t)) continue;
++loaded;
const int before = g_failures;
auto is_lead = [&](unsigned b) {
for (auto& r : t.leads) if (b >= r.first && b <= r.second) return true;
return false;
};
for (unsigned b = 0; b < 256; ++b) {
CHECK(mt_codepage::is_lead_byte(cp, (unsigned char)b) == is_lead(b));
if (is_lead(b)) {
for (unsigned trail = 1; trail < 256; ++trail) {
const unsigned char pair[2] = {(unsigned char)b, (unsigned char)trail};
std::u32string out;
const bool ok = mt_codepage::decode(cp, pair, 2, out);
auto it = t.decode.find(b << 8 | trail);
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
std::fprintf(stderr, "cp%u %02x%02x -> %x, want %x\n", cp, b, trail,
out.empty() ? 0 : unsigned(out[0]), want);
++g_failures;
}
}
} else if (b != 0) {
const unsigned char one[1] = {(unsigned char)b};
std::u32string out;
const bool ok = mt_codepage::decode(cp, one, 1, out);
auto it = t.decode.find(b);
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
std::fprintf(stderr, "cp%u %02x -> %x, want %x\n", cp, b, out.empty() ? 0 : unsigned(out[0]), want);
++g_failures;
}
}
}
for (unsigned wc = 1; wc < 0x10000; ++wc) {
if (wc >= 0xd800 && wc <= 0xdfff) continue;
auto it = t.wctable.find(wc);
bool used = false;
const std::string fit = enc(cp, std::u32string(1, char32_t(wc)), true, &used);
const std::string want_fit = it == t.wctable.end() ? "?" : bytes_of(it->second);
const bool exact = it != t.wctable.end() && t.decode.count(it->second) && t.decode[it->second] == wc;
const std::string strict = enc(cp, std::u32string(1, char32_t(wc)), false);
const std::string want_strict = exact ? bytes_of(it->second) : "?";
if (fit != want_fit || used != (it == t.wctable.end()) || strict != want_strict) {
std::fprintf(stderr, "cp%u U+%04X -> best fit %zu bytes / strict %zu bytes, mismatch\n", cp, wc,
fit.size(), strict.size());
++g_failures;
}
}
std::printf("cp%u: %zu byte sequences, %zu WCTABLE entries: %s\n", cp, t.decode.size(), t.wctable.size(),
g_failures == before ? "ok" : "FAIL");
}
return loaded;
}
int main(int argc, char** argv)
{
spot_checks();
const char* strict = std::getenv("MT_ASSETS_STRICT");
const bool is_strict = strict && std::string(strict) == "1";
const int loaded = argc > 1 ? against_source(argv[1]) : 0;
if (g_failures) {
std::fprintf(stderr, "%d failure(s)\n", g_failures);
return 1;
}
if (loaded < 14) {
std::printf("bestfit sources: %d of 14 found (run script/gen_codepage_tables.py)\n", loaded);
return is_strict ? 1 : 77;
}
std::printf("ok\n");
return 0;
}
+227
View File
@@ -0,0 +1,227 @@
#!/usr/bin/env python3
"""Generate extension/src/codepage/codepage_tables.inc from Microsoft's WindowsBestFit tables.
The port emulates MultiByteToWideChar / WideCharToMultiByte off Windows. Instead of the host iconv
(Android's bionic has no GBK, and iconv tables differ from Windows in the corners), the conversions
use Microsoft's own code page definitions, published by unicode.org:
https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit<cp>.txt
Each file has an MBTABLE (single bytes), DBCSRANGE/DBCSTABLE (lead byte + trail byte) and a WCTABLE
(Unicode -> bytes, including the many-to-one "best fit" entries WideCharToMultiByte uses without
WC_NO_BEST_FIT_CHARS). The generated data keeps the byte -> Unicode side and only the WCTABLE entries
that differ from its first-match inverse, so the runtime can rebuild the encoder.
Usage: python3 script/gen_codepage_tables.py [--cache DIR] [--output FILE]
Downloads missing files into the cache (build/bestfit by default) and checks their SHA-256.
"""
import argparse
import hashlib
import re
import sys
import urllib.request
from pathlib import Path
REPO = Path(__file__).resolve().parent.parent
URL = "https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit{}.txt"
# Every code page 40250 EterLib/Util.cpp names (1252 included; CP_ACP maps to it).
SHA256 = {
874: "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd",
932: "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331",
936: "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae",
949: "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a",
950: "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a",
1250: "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b",
1251: "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7",
1252: "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9",
1253: "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e",
1254: "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a",
1255: "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a",
1256: "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007",
1257: "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366",
1258: "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1",
}
NONE = 0xFFFF
PAIR = re.compile(rb"^\s*0x([0-9a-fA-F]+)\s+0x([0-9a-fA-F]+)")
def fetch(cp: int, cache: Path) -> bytes:
path = cache / f"bestfit{cp}.txt"
if not path.is_file():
cache.mkdir(parents=True, exist_ok=True)
with urllib.request.urlopen(URL.format(cp)) as response:
path.write_bytes(response.read())
data = path.read_bytes()
digest = hashlib.sha256(data).hexdigest()
if digest != SHA256[cp]:
raise SystemExit(f"{path}: SHA-256 {digest} != {SHA256[cp]}")
return data
def parse(cp: int, data: bytes):
single = [NONE] * 256
leads = [] # [(lo, hi)]
dbcs = {} # lead -> {trail: wc}
wctable = {} # wc -> mb
mb_default = None
section = None
lead = None
for raw in data.splitlines():
line = raw.split(b";", 1)[0].strip()
if not line:
continue
head = line.split()[0]
if head == b"CODEPAGE":
if int(line.split()[1]) != cp:
raise SystemExit(f"bestfit{cp}.txt declares {line!r}")
continue
if head == b"CPINFO":
fields = line.split()
if int(fields[2], 16) != 0x3F:
raise SystemExit(f"cp{cp}: byte default char {fields[2]!r} is not '?'")
mb_default = int(fields[3], 16)
continue
if head in (b"MBTABLE", b"WCTABLE", b"DBCSRANGE"):
section = head
continue
if head == b"DBCSTABLE":
section = head
match = re.search(rb"LeadByte\s*=\s*0x([0-9a-fA-F]+)", raw)
lead = int(match.group(1), 16)
dbcs.setdefault(lead, {})
continue
if head == b"ENDCODEPAGE":
break
match = PAIR.match(line)
if not match:
raise SystemExit(f"cp{cp}: unexpected line {raw!r}")
a, b = int(match.group(1), 16), int(match.group(2), 16)
# cp932 lists its second lead range (0xe0-0xfc) after the first range's DBCSTABLEs.
if b"Lead Byte Range" in raw and section in (b"DBCSRANGE", b"DBCSTABLE"):
leads.append((a, b))
continue
if section == b"MBTABLE":
single[a] = b
elif section == b"DBCSRANGE":
leads.append((a, b))
elif section == b"DBCSTABLE":
dbcs[lead][a] = b
elif section == b"WCTABLE":
wctable[a] = b
if mb_default is None:
raise SystemExit(f"cp{cp}: no CPINFO")
for lead in dbcs:
if not any(lo <= lead <= hi for lo, hi in leads):
raise SystemExit(f"cp{cp}: DBCSTABLE 0x{lead:02x} outside the lead ranges {leads}")
return single, leads, dbcs, wctable, mb_default
def decode_map(single, dbcs):
"""bytes -> wc, bytes as int (single byte < 0x100, pair = lead << 8 | trail), ascending."""
out = {}
for b, wc in enumerate(single):
if wc != NONE:
out[b] = wc
for lead in sorted(dbcs):
for trail in sorted(dbcs[lead]):
out[lead << 8 | trail] = dbcs[lead][trail]
return out
def default_inverse(decoded):
"""The encoder the runtime builds: the first byte sequence (ascending) that decodes to wc."""
inverse = {}
for mb in sorted(decoded):
inverse.setdefault(decoded[mb], mb)
return inverse
def encode_overrides(decoded, wctable):
"""WCTABLE entries the default inverse gets wrong: 0 round-trip, 1 best fit, 2 not encodable."""
inverse = default_inverse(decoded)
out = []
for wc in sorted(set(inverse) | set(wctable)):
want = wctable.get(wc)
if want is None:
out.append((wc, 0, 2))
elif decoded.get(want) != wc:
out.append((wc, want, 1))
elif inverse.get(wc) != want:
out.append((wc, want, 0))
return out
def rows(values, per_line=16):
return "\n".join("\t" + ", ".join(values[i:i + per_line]) + "," for i in range(0, len(values), per_line))
def emit(cp, single, leads, dbcs, wctable, mb_default):
decoded = decode_map(single, dbcs)
overrides = encode_overrides(decoded, wctable)
# Sanity: the rebuilt encoder must reproduce WCTABLE exactly.
inverse = default_inverse(decoded)
for wc, mb, kind in overrides:
if kind == 2:
inverse.pop(wc, None)
else:
inverse[wc] = mb
if inverse != wctable:
raise SystemExit(f"cp{cp}: encoder rebuild does not match WCTABLE")
name = f"kCp{cp}"
text = [f"// Code page {cp}: {len(decoded)} byte sequences, {len(wctable)} WCTABLE entries."]
text.append(f"const uint16_t {name}Single[256] = {{\n{rows([f'0x{v:04X}' for v in single])}\n}};")
lead_ranges = "nullptr"
dbcs_rows = "nullptr"
dbcs_data = "nullptr"
if leads:
lead_ranges = f"{name}Leads"
text.append(f"const uint8_t {name}Leads[] = {{" + ", ".join(f"0x{lo:02X}, 0x{hi:02X}" for lo, hi in leads) + "};")
row_defs = []
data = []
for lead in sorted(dbcs):
trails = dbcs[lead]
lo, hi = min(trails), max(trails)
row_defs.append(f"\t{{0x{lead:02X}, 0x{lo:02X}, 0x{hi:02X}, {len(data)}}},")
data.extend(trails.get(t, NONE) for t in range(lo, hi + 1))
dbcs_rows = f"{name}Rows"
dbcs_data = f"{name}Dbcs"
text.append(f"const DbcsRow {name}Rows[] = {{\n" + "\n".join(row_defs) + "\n};")
text.append(f"const uint16_t {name}Dbcs[] = {{\n{rows([f'0x{v:04X}' for v in data])}\n}};")
text.append(f"const EncodeOverride {name}Overrides[] = {{\n"
+ rows([f"{{0x{wc:04X}, 0x{mb:04X}, {kind}}}" for wc, mb, kind in overrides], 6)
+ "\n};" if overrides else f"const EncodeOverride {name}Overrides[] = {{{{0, 0, 2}}}};")
count = f"sizeof({name}Overrides) / sizeof({name}Overrides[0])" if overrides else "0"
entry = (f"\t{{{cp}, 0x{mb_default:04X}, {name}Single, {lead_ranges}, "
f"{len(leads)}, {dbcs_rows}, {len(dbcs) if leads else 0}, {dbcs_data}, "
f"{name}Overrides, {count}}},")
return "\n".join(text) + "\n", entry, len(overrides)
def main():
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--cache", type=Path, default=REPO / "build/bestfit")
parser.add_argument("--output", type=Path, default=REPO / "extension/src/codepage/codepage_tables.inc")
args = parser.parse_args()
body = []
entries = []
for cp in sorted(SHA256):
parsed = parse(cp, fetch(cp, args.cache))
text, entry, n_over = emit(cp, *parsed)
body.append(text)
entries.append(entry)
print(f"cp{cp}: {n_over} encode overrides", file=sys.stderr)
header = ("// Generated by script/gen_codepage_tables.py from Microsoft's WindowsBestFit tables\n"
"// (unicode.org MAPPINGS/VENDORS/MICSFT/WindowsBestFit). Do not edit.\n"
"// 0xFFFF marks a byte / pair with no mapping. Included by codepage.cpp only.\n")
table = "const CodePageTable kTables[] = {\n" + "\n".join(entries) + "\n};\n"
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(header + "\n" + "\n".join(body) + "\n" + table, encoding="ascii")
print(f"wrote {args.output}", file=sys.stderr)
if __name__ == "__main__":
main()