codepage: replace iconv with bundled WindowsBestFit tables

mt_codepage implements code pages 874/932/936/949/950/1250-1258 from
Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py,
SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/
IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build
no longer depends on iconv. port.codepage checks every table against
its source file.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-28 18:45:49 +09:00
co-authored by Claude Opus 5.5
parent 07c101fdc8
commit 0b701d88c3
12 changed files with 7068 additions and 191 deletions
+3 -7
View File
@@ -14,6 +14,8 @@ add_subdirectory(godot-cpp)
# aliases mt3p::sodium / mt3p::zstd / mt3p::minilzo. See docs/THIRD-PARTY.md.
add_subdirectory(third_party)
add_subdirectory(src/codepage)
# --- mtnet: Metin2 net protocol core (no godot-cpp dep; libsodium only) ---
add_library(mtnet STATIC
src/net/text_codec.cpp
@@ -27,13 +29,7 @@ add_library(mtnet STATIC
src/net/classic/classic_cipher.cpp
)
target_include_directories(mtnet PUBLIC src/net)
target_link_libraries(mtnet PUBLIC mt3p::sodium mt3p::minilzo mt3p::cryptopp)
# macOS ships iconv as a system library rather than libc. Linux/glibc and
# Windows use the platform implementation without this extra link item.
find_library(MT_ICONV_LIBRARY NAMES iconv)
if(MT_ICONV_LIBRARY)
target_link_libraries(mtnet PUBLIC ${MT_ICONV_LIBRARY})
endif()
target_link_libraries(mtnet PUBLIC mt3p::sodium mt3p::minilzo mt3p::cryptopp mt_codepage)
target_compile_features(mtnet PUBLIC cxx_std_20)
# §1.5: non-EUROPE CG_CLIENT_VERSION timestamp, in C __TIMESTAMP__ form
# ("Www Mmm dd hh:mm:ss yyyy"); regenerated on each CMake configure.
+8
View File
@@ -0,0 +1,8 @@
# mt_codepage — Windows ANSI code pages (874/932/936/949/950/1250-1258) from Microsoft's WindowsBestFit
# tables; codepage_tables.inc is generated by script/gen_codepage_tables.py. Used by port_logic
# (Win32Crt MultiByteToWideChar / WideCharToMultiByte) and mtnet (40250 CP936 wire text), in place of
# the host iconv.
add_library(mt_codepage STATIC codepage.cpp)
target_include_directories(mt_codepage PUBLIC ${CMAKE_CURRENT_SOURCE_DIR})
target_compile_features(mt_codepage PUBLIC cxx_std_17)
set_target_properties(mt_codepage PROPERTIES POSITION_INDEPENDENT_CODE ON)
+169
View File
@@ -0,0 +1,169 @@
#include "codepage.h"
#include <algorithm>
#include <cstdint>
#include <memory>
#include <mutex>
namespace mt_codepage {
namespace {
constexpr uint16_t kNone = 0xFFFF;
struct DbcsRow {
uint8_t lead;
uint8_t trail_lo;
uint8_t trail_hi;
uint32_t offset; // into the Dbcs array, trail_lo first
};
// A WCTABLE entry the first-match inverse of the byte tables gets wrong.
struct EncodeOverride {
uint16_t wc;
uint16_t mb;
uint8_t kind; // 0 exact (round-trips), 1 best fit only, 2 not encodable
};
struct CodePageTable {
unsigned cp;
uint16_t mb_default; // Unicode char for an unmapped byte / pair (CPINFO)
const uint16_t* single;
const uint8_t* leads; // lo, hi pairs
size_t lead_count;
const DbcsRow* rows;
size_t row_count;
const uint16_t* dbcs;
const EncodeOverride* overrides;
size_t override_count;
};
#include "codepage_tables.inc"
constexpr size_t kTableCount = sizeof(kTables) / sizeof(kTables[0]);
const CodePageTable* find(unsigned cp)
{
for (const CodePageTable& t : kTables)
if (t.cp == cp) return &t;
return nullptr;
}
bool lead(const CodePageTable& t, unsigned char b)
{
for (size_t i = 0; i < t.lead_count; ++i)
if (b >= t.leads[i * 2] && b <= t.leads[i * 2 + 1]) return true;
return false;
}
uint16_t pair(const CodePageTable& t, unsigned char l, unsigned char trail)
{
for (size_t i = 0; i < t.row_count; ++i) {
const DbcsRow& r = t.rows[i];
if (r.lead != l) continue;
if (trail < r.trail_lo || trail > r.trail_hi) return kNone;
return t.dbcs[r.offset + (trail - r.trail_lo)];
}
return kNone;
}
// Exact encoder (BMP code point -> bytes, kNone when unmapped), built on first use per code page.
struct Encoder {
std::once_flag once;
std::unique_ptr<uint16_t[]> exact;
};
Encoder g_encoders[kTableCount];
const uint16_t* exact_encoder(const CodePageTable& t)
{
Encoder& e = g_encoders[&t - kTables];
std::call_once(e.once, [&] {
e.exact.reset(new uint16_t[0x10000]);
std::fill_n(e.exact.get(), 0x10000, kNone);
uint16_t* map = e.exact.get();
// First match in ascending byte order, as the generator assumed.
for (unsigned b = 0; b < 256; ++b)
if (t.single[b] != kNone && map[t.single[b]] == kNone) map[t.single[b]] = uint16_t(b);
for (size_t i = 0; i < t.row_count; ++i) {
const DbcsRow& r = t.rows[i];
for (unsigned trail = r.trail_lo; trail <= r.trail_hi; ++trail) {
const uint16_t wc = t.dbcs[r.offset + (trail - r.trail_lo)];
if (wc != kNone && map[wc] == kNone) map[wc] = uint16_t(r.lead << 8 | trail);
}
}
for (size_t i = 0; i < t.override_count; ++i) {
const EncodeOverride& o = t.overrides[i];
if (o.kind == 0) map[o.wc] = o.mb;
else if (o.kind == 2) map[o.wc] = kNone;
}
});
return e.exact.get();
}
uint16_t best_fit(const CodePageTable& t, char32_t c)
{
const EncodeOverride* end = t.overrides + t.override_count;
const EncodeOverride* it = std::lower_bound(t.overrides, end, c,
[](const EncodeOverride& o, char32_t v) { return o.wc < v; });
return it != end && it->wc == c && it->kind == 1 ? it->mb : kNone;
}
} // namespace
bool supported(unsigned cp)
{
return find(cp) != nullptr;
}
bool is_lead_byte(unsigned cp, unsigned char b)
{
const CodePageTable* t = find(cp);
return t && lead(*t, b);
}
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out)
{
const CodePageTable* t = find(cp);
if (!t) return false;
bool ok = true;
for (size_t i = 0; i < n;) {
const unsigned char b = p[i];
uint16_t wc = kNone;
if (lead(*t, b) && i + 1 < n && p[i + 1] != 0) {
wc = pair(*t, b, p[i + 1]);
i += 2;
} else {
wc = t->single[b];
++i;
}
if (wc == kNone) {
wc = t->mb_default;
ok = false;
}
out.push_back(wc);
}
return ok;
}
void encode(unsigned cp, std::u32string_view in, std::string& out, bool allow_best_fit, bool* used_default)
{
const CodePageTable* t = find(cp);
const uint16_t* exact = t ? exact_encoder(*t) : nullptr;
for (char32_t c : in) {
uint16_t mb = kNone;
if (exact && c < 0x10000) {
mb = exact[c];
if (mb == kNone && allow_best_fit) mb = best_fit(*t, c);
}
if (mb == kNone) {
out.push_back('?');
if (used_default) *used_default = true;
} else if (mb > 0xFF) {
out.push_back(char(mb >> 8));
out.push_back(char(mb & 0xFF));
} else {
out.push_back(char(mb));
}
}
}
} // namespace mt_codepage
+29
View File
@@ -0,0 +1,29 @@
#pragma once
// Windows ANSI code pages without the host iconv: MultiByteToWideChar / WideCharToMultiByte semantics
// driven by Microsoft's WindowsBestFit tables (script/gen_codepage_tables.py). One implementation for
// macOS, Linux, Android and iOS, so every platform decodes the 40250 locale and wire text the same way.
//
// Covered: 874, 932, 936, 949, 950, 1250-1258 (every code page 40250 EterLib/Util.cpp names).
// UTF-8 is not handled here.
#include <cstddef>
#include <string>
#include <string_view>
namespace mt_codepage {
bool supported(unsigned cp);
// Byte string -> code points, as MultiByteToWideChar. A lead byte takes the next non-NUL byte as its
// trail; an unmapped byte or pair becomes the code page's default char (U+30FB for 932, else '?').
// Returns false when a default char was substituted (MB_ERR_INVALID_CHARS would fail).
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out);
// Code points -> bytes, as WideCharToMultiByte. With best_fit (no WC_NO_BEST_FIT_CHARS) a character
// without an exact mapping may take its WCTABLE best fit ("Ā" -> "A" on 1252); otherwise, and for
// anything unmapped, '?' is written and *used_default set.
void encode(unsigned cp, std::u32string_view in, std::string& out, bool best_fit, bool* used_default);
bool is_lead_byte(unsigned cp, unsigned char b);
} // namespace mt_codepage
File diff suppressed because it is too large Load Diff
+52 -57
View File
@@ -1,16 +1,12 @@
#include "text_codec.h"
#include <algorithm>
#include <cerrno>
#include <cstring>
#if defined(_WIN32)
#include <windows.h>
#elif defined(__has_include)
#if __has_include(<iconv.h>)
#include <iconv.h>
#define MT_TEXT_CODEC_HAS_ICONV 1
#endif
#else
// CP936 from Microsoft's own table, not the host iconv (bionic has no GBK).
#include "codepage.h"
#endif
namespace mtnet {
@@ -69,61 +65,66 @@ bool convert_from_windows(std::string_view input, std::string &out) {
return WideCharToMultiByte(CP_UTF8, 0, wide.data(), wide_len, out.data(), utf8_len,
nullptr, nullptr) == utf8_len;
}
#elif defined(MT_TEXT_CODEC_HAS_ICONV)
bool convert_iconv(std::string_view input, const char *from, const char *to, std::string &out) {
iconv_t cd = iconv_open(to, from);
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
std::string buffer(std::max<size_t>(input.size() * 4 + 16, 32), '\0');
char *in_ptr = const_cast<char *>(input.data());
size_t in_left = input.size();
char *out_ptr = buffer.data();
size_t out_left = buffer.size();
while (in_left != 0) {
const size_t rc = iconv(cd, &in_ptr, &in_left, &out_ptr, &out_left);
if (rc == static_cast<size_t>(-1)) {
iconv_close(cd);
#else
// Mirrors the Windows branch: CP936 without best-fit substitutions, and a strict decode.
constexpr unsigned kWireCodePage = 936;
bool decode_utf8_strict(std::string_view input, std::u32string &out) {
for (size_t i = 0; i < input.size();) {
const size_t seq = utf8_sequence_length(input.data() + i, input.size() - i);
if (seq == 0) return false;
const unsigned char c = static_cast<unsigned char>(input[i]);
char32_t cp = seq == 1 ? c : seq == 2 ? (c & 0x1F) : seq == 3 ? (c & 0x0F) : (c & 0x07);
for (size_t k = 1; k < seq; ++k) {
const unsigned char t = static_cast<unsigned char>(input[i + k]);
if ((t & 0xC0) != 0x80) return false;
cp = (cp << 6) | (t & 0x3F);
}
if ((seq == 3 && cp < 0x800) || (seq == 4 && (cp < 0x10000 || cp > 0x10FFFF)) ||
(cp >= 0xD800 && cp <= 0xDFFF)) {
return false;
}
out.push_back(cp);
i += seq;
}
// Flush stateful encodings, even though GB2312 itself is stateless.
iconv(cd, nullptr, nullptr, &out_ptr, &out_left);
iconv_close(cd);
out.assign(buffer.data(), static_cast<size_t>(out_ptr - buffer.data()));
return true;
}
void append_utf8(char32_t cp, std::string &out) {
if (cp < 0x80) {
out.push_back(static_cast<char>(cp));
} else if (cp < 0x800) {
out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
} else if (cp < 0x10000) {
out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
} else {
out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
}
}
bool encode_one(std::string_view input, std::string &out) {
return convert_iconv(input, "UTF-8", "GB2312", out) ||
convert_iconv(input, "UTF-8", "GBK", out);
std::u32string wide;
if (!decode_utf8_strict(input, wide)) return false;
bool used_default = false;
out.clear();
mt_codepage::encode(kWireCodePage, wide, out, false, &used_default);
return !used_default;
}
bool decode_wire(std::string_view input, std::string &out) {
return convert_iconv(input, "GB2312", "UTF-8", out) ||
convert_iconv(input, "GBK", "UTF-8", out);
}
#else
bool encode_one(std::string_view input, std::string &out) {
// A platform without either Windows code-page APIs or iconv must never
// send UTF-8 bytes into a 40250 fixed legacy field. ASCII is lossless;
// non-ASCII is reported as unrepresentable so wire_text_fits() rejects it
// and to_wire() can safely substitute '?' rather than leaking a 3-byte
// UTF-8 sequence into a 24-byte name field.
const size_t n = utf8_byte_truncate(input.data(), input.size(), input.size());
if (n != input.size()) return false;
for (unsigned char c : input) {
if (c >= 0x80) return false;
std::u32string wide;
if (!mt_codepage::decode(kWireCodePage, reinterpret_cast<const unsigned char *>(input.data()),
input.size(), wide)) {
return false;
}
out.assign(input.data(), input.size());
return true;
}
bool decode_wire(std::string_view input, std::string &out) {
// The fallback decoder is deliberately conservative for the same reason:
// returning raw high-bit bytes would create invalid UTF-8 in the client.
for (unsigned char c : input) {
if (c >= 0x80) return false;
}
out.assign(input.data(), input.size());
out.clear();
for (char32_t c : wide) append_utf8(c, out);
return true;
}
#endif
@@ -176,7 +177,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
size_t n = 0;
while (n < cap && bytes[n] != '\0') ++n;
std::string decoded;
#if defined(_WIN32) || defined(MT_TEXT_CODEC_HAS_ICONV)
// A fixed field can arrive without a terminator and a caller may provide a
// larger view than the actual field. If that view ends on a legacy lead byte,
// drop only the malformed suffix and preserve the complete preceding text.
@@ -188,11 +188,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
#endif
if (ok) return decoded;
}
#else
if (decode_wire(std::string_view(bytes, n), decoded)) {
return decoded;
}
#endif
// Keep malformed input visible without leaking legacy bytes as invalid UTF-8.
decoded.clear();
for (size_t i = 0; i < n; ++i) {
+13 -5
View File
@@ -31,12 +31,12 @@ target_include_directories(port_logic PUBLIC ${MT_PORT_SHIMS})
# Libraries 40250 links that are vendored as-is: <lzo/lzo1x.h> (EterBase/lzo.h), <cryptopp/*>
# (EterBase/cipher.h), <Python-2.7/*> (ScriptLib).
target_link_libraries(port_logic PUBLIC mt3p::minilzo mt3p::cryptopp)
# common/Win32Crt.cpp converts non-UTF-8 code pages with iconv. macOS ships it as a separate library;
# extension/CMakeLists.txt finds it for mtnet, but the port-only build (script/port_gate.sh) never reads that file.
find_library(MT_ICONV_LIBRARY NAMES iconv)
if(MT_ICONV_LIBRARY)
target_link_libraries(port_logic PUBLIC ${MT_ICONV_LIBRARY})
# common/Win32Crt.cpp converts ANSI code pages with mt_codepage. extension/CMakeLists.txt adds it for
# mtnet; the port-only build (script/port_gate.sh, native_render) never reads that file.
if(NOT TARGET mt_codepage)
add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/../codepage ${CMAKE_CURRENT_BINARY_DIR}/../codepage)
endif()
target_link_libraries(port_logic PUBLIC mt_codepage)
if(TARGET mtpython)
target_link_libraries(port_logic PUBLIC mt3p::python)
endif()
@@ -173,6 +173,14 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})
set_tests_properties(port.eterpack PROPERTIES SKIP_RETURN_CODE 77)
# mt_codepage against Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py caches them
# in build/bestfit); skipped (77) without them unless MT_ASSETS_STRICT=1.
add_executable(port_codepage_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/codepage_test.cpp)
target_link_libraries(port_codepage_test PRIVATE mt_codepage)
add_test(NAME port.codepage COMMAND $<TARGET_FILE:port_codepage_test>
${CMAKE_CURRENT_SOURCE_DIR}/../../../build/bestfit)
set_tests_properties(port.codepage PROPERTIES SKIP_RETURN_CODE 77)
# 批次 2V0-g: the 40250 font path over the FreeType GDI emulation; needs a system Tahoma.
add_executable(port_text_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/port_text_test.cpp)
target_link_libraries(port_text_test PRIVATE port_platform)
+37 -118
View File
@@ -11,11 +11,11 @@
#include <thread>
#include <vector>
#if __has_include(<iconv.h>)
#include <iconv.h>
#define MT_WIN32CRT_HAS_ICONV 1
#endif
#include <glob.h>
// ANSI code pages come from Microsoft's own tables, not the host iconv (bionic has no GBK).
#include "codepage.h"
#include <algorithm>
#include <dirent.h>
#include <fnmatch.h>
#include <sys/stat.h>
#include <unistd.h>
@@ -103,17 +103,30 @@ HANDLE FindFirstFile(const char* pattern, WIN32_FIND_DATA* data)
std::string normalized(pattern);
for (char& ch : normalized)
if (ch == '\\') ch = '/';
glob_t matches{};
const int result = glob(normalized.c_str(), 0, nullptr, &matches);
if (result != 0 || matches.gl_pathc == 0)
// Win32 wildcards only appear in the last component; match it against the directory like
// glob(3) would (sorted, '*' skips dot files). glob itself is API 28+ on Android.
std::vector<std::string> paths;
const size_t slash = normalized.rfind('/');
const std::string dir = slash == std::string::npos ? std::string() : normalized.substr(0, slash + 1);
const std::string name = normalized.substr(dir.size());
if (name.find_first_of("*?[") == std::string::npos)
{
globfree(&matches);
return INVALID_HANDLE_VALUE;
struct stat st;
if (::stat(normalized.c_str(), &st) == 0)
paths.push_back(normalized);
}
else if (DIR* d = ::opendir(dir.empty() ? "." : dir.c_str()))
{
while (const dirent* entry = ::readdir(d))
if (::fnmatch(name.c_str(), entry->d_name, FNM_PERIOD) == 0)
paths.push_back(dir + entry->d_name);
::closedir(d);
std::sort(paths.begin(), paths.end());
}
if (paths.empty())
return INVALID_HANDLE_VALUE;
auto* state = new FindState;
for (size_t i = 0; i < matches.gl_pathc; ++i)
state->paths.emplace_back(matches.gl_pathv[i]);
globfree(&matches);
state->paths = std::move(paths);
fill_find_data(state->paths[0], data);
state->next = 1;
return state;
@@ -199,13 +212,8 @@ DWORD GetTickCount()
namespace {
// Windows-1252 0x80..0x9F; the other bytes map to the same code point (0 marks the undefined ones).
const char32_t kCp1252High[32] = {
0x20AC, 0, 0x201A, 0x0192, 0x201E, 0x2026, 0x2020, 0x2021, 0x02C6, 0x2030, 0x0160, 0x2039, 0x0152, 0, 0x017D, 0,
0, 0x2018, 0x2019, 0x201C, 0x201D, 0x2022, 0x2013, 0x2014, 0x02DC, 0x2122, 0x0161, 0x203A, 0x0153, 0, 0x017E, 0x0178,
};
bool is_cp1252(UINT cp) { return cp == CP_ACP || cp == 1252; }
// CP_ACP is the Western system code page here, as on the 40250 target machines.
UINT ansi(UINT cp) { return cp == CP_ACP ? 1252 : cp; }
bool decode_utf8(const unsigned char* p, size_t n, std::u32string& out)
{
@@ -237,108 +245,25 @@ void encode_utf8(char32_t cp, std::string& out)
else { out.push_back(char(0xF0 | (cp >> 18))); out.push_back(char(0x80 | ((cp >> 12) & 0x3F))); out.push_back(char(0x80 | ((cp >> 6) & 0x3F))); out.push_back(char(0x80 | (cp & 0x3F))); }
}
#if defined(MT_WIN32CRT_HAS_ICONV)
std::string iconv_name(UINT cp)
{
switch (cp) {
case 949: return "CP949";
case 936: return "GBK";
case 950: return "BIG5";
case 932: return "SHIFT_JIS";
case 874: return "CP874";
default: return "CP" + std::to_string(cp);
}
}
// One character at a time, so an undecodable byte becomes one default char like Windows does.
bool iconv_decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
{
iconv_t cd = iconv_open("UTF-32LE", iconv_name(cp).c_str());
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
for (size_t i = 0; i < n;) {
unsigned char wide[16];
bool done = false;
for (size_t len = 1; len <= 4 && i + len <= n && !done; ++len) {
iconv(cd, nullptr, nullptr, nullptr, nullptr);
char* in = const_cast<char*>(reinterpret_cast<const char*>(p + i));
size_t in_left = len;
char* o = reinterpret_cast<char*>(wide);
size_t o_left = sizeof(wide);
if (iconv(cd, &in, &in_left, &o, &o_left) != size_t(-1) && in_left == 0 && o_left < sizeof(wide)) {
for (size_t k = 0; k + 4 <= sizeof(wide) - o_left; k += 4)
out.push_back(char32_t(wide[k] | (wide[k + 1] << 8) | (wide[k + 2] << 16) | (unsigned(wide[k + 3]) << 24)));
i += len;
done = true;
}
}
if (!done) { out.push_back(0xFFFD); ++i; }
}
iconv_close(cd);
return true;
}
bool iconv_encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
{
iconv_t cd = iconv_open(iconv_name(cp).c_str(), "UTF-32LE");
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
for (char32_t c : in) {
unsigned char src[4] = {static_cast<unsigned char>(c), static_cast<unsigned char>(c >> 8), static_cast<unsigned char>(c >> 16), static_cast<unsigned char>(c >> 24)};
char buf[16];
char* ip = reinterpret_cast<char*>(src);
size_t il = 4;
char* op = buf;
size_t ol = sizeof(buf);
iconv(cd, nullptr, nullptr, nullptr, nullptr);
if (iconv(cd, &ip, &il, &op, &ol) != size_t(-1) && il == 0) {
out.append(buf, sizeof(buf) - ol);
} else {
out.push_back('?');
*used_default = true;
}
}
iconv_close(cd);
return true;
}
#endif
bool decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
{
if (cp == CP_UTF8) return decode_utf8(p, n, out);
if (is_cp1252(cp)) {
for (size_t i = 0; i < n; ++i) {
const unsigned char c = p[i];
out.push_back(c >= 0x80 && c < 0xA0 ? (kCp1252High[c - 0x80] ? kCp1252High[c - 0x80] : char32_t(c)) : char32_t(c));
}
return true;
}
#if defined(MT_WIN32CRT_HAS_ICONV)
if (iconv_decode(cp, p, n, out)) return true;
#endif
// No converter for this code page: ASCII passes, the rest becomes U+FFFD.
if (mt_codepage::supported(ansi(cp))) return mt_codepage::decode(ansi(cp), p, n, out);
// No table for this code page: ASCII passes, the rest becomes U+FFFD.
for (size_t i = 0; i < n; ++i) out.push_back(p[i] < 0x80 ? char32_t(p[i]) : char32_t(0xFFFD));
return true;
}
void encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
void encode(UINT cp, DWORD flags, const std::u32string& in, std::string& out, bool* used_default)
{
if (cp == CP_UTF8) {
for (char32_t c : in) encode_utf8(c, out);
return;
}
if (is_cp1252(cp)) {
for (char32_t c : in) {
if (c < 0x80 || (c >= 0xA0 && c < 0x100)) { out.push_back(char(c)); continue; }
int hit = -1;
for (int k = 0; k < 32; ++k)
if (kCp1252High[k] == c) { hit = k; break; }
if (hit >= 0) out.push_back(char(0x80 + hit));
else { out.push_back('?'); *used_default = true; }
}
if (mt_codepage::supported(ansi(cp))) {
mt_codepage::encode(ansi(cp), in, out, (flags & WC_NO_BEST_FIT_CHARS) == 0, used_default);
return;
}
#if defined(MT_WIN32CRT_HAS_ICONV)
if (iconv_encode(cp, in, out, used_default)) return;
#endif
for (char32_t c : in) {
if (c < 0x80) out.push_back(char(c));
else { out.push_back('?'); *used_default = true; }
@@ -361,7 +286,7 @@ int MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int
return int(wide.size());
}
int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWideChar,
int WideCharToMultiByte(UINT CodePage, DWORD dwFlags, LPCWSTR lpWideCharStr, int cchWideChar,
LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar, LPBOOL lpUsedDefaultChar)
{
if (!lpWideCharStr || cchWideChar == 0 || cbMultiByte < 0) return 0;
@@ -372,14 +297,14 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
for (size_t i = 0; i < n; ++i) wide[i] = char32_t(lpWideCharStr[i]);
std::string bytes;
bool used_default = false;
encode(CodePage, wide, bytes, &used_default);
encode(CodePage, dwFlags, wide, bytes, &used_default);
if (used_default && lpDefaultChar && CodePage != CP_UTF8) {
// '?' was the placeholder; the caller's default char replaces it only where a fallback happened.
std::string redo;
for (char32_t c : wide) {
std::string one;
bool miss = false;
encode(CodePage, std::u32string(1, c), one, &miss);
encode(CodePage, dwFlags, std::u32string(1, c), one, &miss);
redo += miss ? std::string(1, *lpDefaultChar) : one;
}
bytes.swap(redo);
@@ -393,13 +318,7 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
BOOL IsDBCSLeadByteEx(UINT CodePage, BYTE TestChar)
{
switch (CodePage) {
case 932: return (TestChar >= 0x81 && TestChar <= 0x9F) || (TestChar >= 0xE0 && TestChar <= 0xFC);
case 936:
case 949:
case 950: return TestChar >= 0x81 && TestChar <= 0xFE;
default: return FALSE;
}
return mt_codepage::is_lead_byte(ansi(CodePage), TestChar) ? TRUE : FALSE;
}
LPSTR CharNextExA(WORD CodePage, LPCSTR lpCurrentChar, DWORD)
+2 -2
View File
@@ -104,8 +104,8 @@ void DeleteCriticalSection(LPCRITICAL_SECTION cs);
void EnterCriticalSection(LPCRITICAL_SECTION cs);
void LeaveCriticalSection(LPCRITICAL_SECTION cs);
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 and
// CP_ACP/1252 are converted directly; other code pages go through iconv where the platform has it.
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 is converted
// directly; CP_ACP (= 1252) and the other ANSI code pages use Microsoft's tables (codepage/codepage.h).
#define CP_ACP 0
#define CP_UTF8 65001
#define MB_PRECOMPOSED 0x00000001
+214
View File
@@ -0,0 +1,214 @@
// codepage_test — mt_codepage against Microsoft's WindowsBestFit source tables, plus spot checks of the
// characters each shipped 40250 locale needs.
//
// Argument: the directory holding bestfit<cp>.txt (script/gen_codepage_tables.py downloads them to
// build/bestfit). Without it only the spot checks run and the test exits 77 (ctest SKIP), unless
// MT_ASSETS_STRICT=1.
#include "codepage.h"
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <fstream>
#include <map>
#include <sstream>
#include <string>
#include <vector>
static int g_failures = 0;
#define CHECK(cond) \
do { \
if (!(cond)) { \
std::fprintf(stderr, "%s:%d: CHECK(%s)\n", __FILE__, __LINE__, #cond); \
++g_failures; \
} \
} while (0)
static std::u32string dec(unsigned cp, const char* bytes, bool* ok = nullptr)
{
std::u32string out;
const bool r = mt_codepage::decode(cp, reinterpret_cast<const unsigned char*>(bytes), std::strlen(bytes), out);
if (ok) *ok = r;
return out;
}
static std::string enc(unsigned cp, std::u32string in, bool best_fit = false, bool* used_default = nullptr)
{
std::string out;
bool used = false;
mt_codepage::encode(cp, in, out, best_fit, &used);
if (used_default) *used_default = used;
return out;
}
static void spot_checks()
{
// zh (936): "中文" D6D0 CEC4, and the Windows-only single byte 0x80 = euro.
CHECK(dec(936, "\xd6\xd0\xce\xc4") == U"中文");
CHECK(enc(936, U"中文") == "\xd6\xd0\xce\xc4");
CHECK(dec(936, "\x80") == U"€");
// GBK extension beyond GB2312 (U+4E02 = 81 40), both directions.
CHECK(dec(936, "\x81\x40") == U"丂");
CHECK(enc(936, U"丂") == "\x81\x40");
// A lead byte with nothing after it is one default char and reports failure.
bool ok = true;
CHECK(dec(936, "a\xd6", &ok) == U"a?" && !ok);
// An unmapped pair is consumed whole (932 uses U+30FB as its default).
CHECK(dec(932, "\x85\x40") == U"・");
CHECK(dec(932, "\x82\xa0") == U"あ");
// The shipped European locales.
CHECK(dec(1252, "\x80\xe9") == U"€é"); // en/de/fr/...: euro, é
CHECK(dec(1252, "\x81") == U"\u0081"); // undefined byte keeps its value, as Windows
CHECK(dec(1250, "\xb9\xb3\x9c") == U"ąłś"); // pl/cz/hu/ro: ą ł ś
CHECK(dec(1251, "\xc6\xe8") == U"Жи"); // ru: Жи
CHECK(dec(1253, "\xd9\xe1") == U"Ωα"); // gr: Ωα
CHECK(dec(1254, "\xf0\xfe\xdd") == U"ğşİ"); // tr: ğ ş İ
CHECK(enc(1250, U"ąłś") == "\xb9\xb3\x9c");
CHECK(enc(1251, U"Жи") == "\xc6\xe8");
CHECK(enc(1253, U"Ωα") == "\xd9\xe1");
CHECK(enc(1254, U"ğşİ") == "\xf0\xfe\xdd");
// Best fit only when WC_NO_BEST_FIT_CHARS is absent: 1252 has no "Ā", its best fit is "A".
bool used = false;
CHECK(enc(1252, U"Ā", true, &used) == "A" && !used);
CHECK(enc(1252, U"Ā", false, &used) == "?" && used);
// Unmapped either way; beyond the BMP too.
CHECK(enc(1251, U"中", true, &used) == "?" && used);
CHECK(enc(936, U"\U0001F600", true, &used) == "?" && used);
CHECK(mt_codepage::is_lead_byte(936, 0x81) && !mt_codepage::is_lead_byte(936, 0x80));
CHECK(mt_codepage::is_lead_byte(932, 0xe0) && !mt_codepage::is_lead_byte(932, 0xa1));
CHECK(!mt_codepage::is_lead_byte(1250, 0xb9));
CHECK(!mt_codepage::supported(65001) && !mt_codepage::supported(437));
}
struct SourceTable {
unsigned mb_default = 0x3f;
std::map<unsigned, unsigned> decode; // bytes (pair = lead << 8 | trail) -> wc
std::map<unsigned, unsigned> wctable;
std::vector<std::pair<unsigned, unsigned>> leads;
};
static bool load(const std::string& path, SourceTable& t)
{
std::ifstream in(path, std::ios::binary);
if (!in) return false;
std::string raw, section;
unsigned lead = 0;
while (std::getline(in, raw)) {
std::string line = raw.substr(0, raw.find(';'));
std::istringstream words(line);
std::string head;
if (!(words >> head)) continue;
if (head == "CPINFO") {
std::string size, byte_default, wc_default;
words >> size >> byte_default >> wc_default;
t.mb_default = unsigned(std::stoul(wc_default, nullptr, 16));
continue;
}
if (head == "MBTABLE" || head == "WCTABLE" || head == "DBCSRANGE") { section = head; continue; }
if (head == "DBCSTABLE") {
section = head;
lead = unsigned(std::stoul(raw.substr(raw.find("0x", raw.find("LeadByte"))), nullptr, 16));
continue;
}
if (head == "ENDCODEPAGE") break;
if (head.rfind("0x", 0) != 0) continue;
std::string second;
words >> second;
const unsigned a = unsigned(std::stoul(head, nullptr, 16));
const unsigned b = unsigned(std::stoul(second, nullptr, 16));
if (raw.find("Lead Byte Range") != std::string::npos) t.leads.push_back({a, b});
else if (section == "MBTABLE") t.decode[a] = b;
else if (section == "DBCSTABLE") t.decode[lead << 8 | a] = b;
else if (section == "WCTABLE") t.wctable[a] = b;
}
return !t.decode.empty() && !t.wctable.empty();
}
static std::string bytes_of(unsigned mb)
{
return mb > 0xff ? std::string{char(mb >> 8), char(mb & 0xff)} : std::string(1, char(mb));
}
// Every byte / pair the source defines decodes to it, every other one to the default char; every
// WCTABLE entry encodes to it with best fit, and without best fit only when it round-trips.
static int against_source(const std::string& dir)
{
const unsigned cps[] = {874, 932, 936, 949, 950, 1250, 1251, 1252, 1253, 1254, 1255, 1256, 1257, 1258};
int loaded = 0;
for (unsigned cp : cps) {
SourceTable t;
if (!load(dir + "/bestfit" + std::to_string(cp) + ".txt", t)) continue;
++loaded;
const int before = g_failures;
auto is_lead = [&](unsigned b) {
for (auto& r : t.leads) if (b >= r.first && b <= r.second) return true;
return false;
};
for (unsigned b = 0; b < 256; ++b) {
CHECK(mt_codepage::is_lead_byte(cp, (unsigned char)b) == is_lead(b));
if (is_lead(b)) {
for (unsigned trail = 1; trail < 256; ++trail) {
const unsigned char pair[2] = {(unsigned char)b, (unsigned char)trail};
std::u32string out;
const bool ok = mt_codepage::decode(cp, pair, 2, out);
auto it = t.decode.find(b << 8 | trail);
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
std::fprintf(stderr, "cp%u %02x%02x -> %x, want %x\n", cp, b, trail,
out.empty() ? 0 : unsigned(out[0]), want);
++g_failures;
}
}
} else if (b != 0) {
const unsigned char one[1] = {(unsigned char)b};
std::u32string out;
const bool ok = mt_codepage::decode(cp, one, 1, out);
auto it = t.decode.find(b);
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
std::fprintf(stderr, "cp%u %02x -> %x, want %x\n", cp, b, out.empty() ? 0 : unsigned(out[0]), want);
++g_failures;
}
}
}
for (unsigned wc = 1; wc < 0x10000; ++wc) {
if (wc >= 0xd800 && wc <= 0xdfff) continue;
auto it = t.wctable.find(wc);
bool used = false;
const std::string fit = enc(cp, std::u32string(1, char32_t(wc)), true, &used);
const std::string want_fit = it == t.wctable.end() ? "?" : bytes_of(it->second);
const bool exact = it != t.wctable.end() && t.decode.count(it->second) && t.decode[it->second] == wc;
const std::string strict = enc(cp, std::u32string(1, char32_t(wc)), false);
const std::string want_strict = exact ? bytes_of(it->second) : "?";
if (fit != want_fit || used != (it == t.wctable.end()) || strict != want_strict) {
std::fprintf(stderr, "cp%u U+%04X -> best fit %zu bytes / strict %zu bytes, mismatch\n", cp, wc,
fit.size(), strict.size());
++g_failures;
}
}
std::printf("cp%u: %zu byte sequences, %zu WCTABLE entries: %s\n", cp, t.decode.size(), t.wctable.size(),
g_failures == before ? "ok" : "FAIL");
}
return loaded;
}
int main(int argc, char** argv)
{
spot_checks();
const char* strict = std::getenv("MT_ASSETS_STRICT");
const bool is_strict = strict && std::string(strict) == "1";
const int loaded = argc > 1 ? against_source(argv[1]) : 0;
if (g_failures) {
std::fprintf(stderr, "%d failure(s)\n", g_failures);
return 1;
}
if (loaded < 14) {
std::printf("bestfit sources: %d of 14 found (run script/gen_codepage_tables.py)\n", loaded);
return is_strict ? 1 : 77;
}
std::printf("ok\n");
return 0;
}