codepage: replace iconv with bundled WindowsBestFit tables
mt_codepage implements code pages 874/932/936/949/950/1250-1258 from Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py, SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/ IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build no longer depends on iconv. port.codepage checks every table against its source file. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
07c101fdc8
commit
0b701d88c3
@@ -0,0 +1,8 @@
|
||||
# mt_codepage — Windows ANSI code pages (874/932/936/949/950/1250-1258) from Microsoft's WindowsBestFit
|
||||
# tables; codepage_tables.inc is generated by script/gen_codepage_tables.py. Used by port_logic
|
||||
# (Win32Crt MultiByteToWideChar / WideCharToMultiByte) and mtnet (40250 CP936 wire text), in place of
|
||||
# the host iconv.
|
||||
add_library(mt_codepage STATIC codepage.cpp)
|
||||
target_include_directories(mt_codepage PUBLIC ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
target_compile_features(mt_codepage PUBLIC cxx_std_17)
|
||||
set_target_properties(mt_codepage PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
@@ -0,0 +1,169 @@
|
||||
#include "codepage.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
|
||||
namespace mt_codepage {
|
||||
namespace {
|
||||
|
||||
constexpr uint16_t kNone = 0xFFFF;
|
||||
|
||||
struct DbcsRow {
|
||||
uint8_t lead;
|
||||
uint8_t trail_lo;
|
||||
uint8_t trail_hi;
|
||||
uint32_t offset; // into the Dbcs array, trail_lo first
|
||||
};
|
||||
|
||||
// A WCTABLE entry the first-match inverse of the byte tables gets wrong.
|
||||
struct EncodeOverride {
|
||||
uint16_t wc;
|
||||
uint16_t mb;
|
||||
uint8_t kind; // 0 exact (round-trips), 1 best fit only, 2 not encodable
|
||||
};
|
||||
|
||||
struct CodePageTable {
|
||||
unsigned cp;
|
||||
uint16_t mb_default; // Unicode char for an unmapped byte / pair (CPINFO)
|
||||
const uint16_t* single;
|
||||
const uint8_t* leads; // lo, hi pairs
|
||||
size_t lead_count;
|
||||
const DbcsRow* rows;
|
||||
size_t row_count;
|
||||
const uint16_t* dbcs;
|
||||
const EncodeOverride* overrides;
|
||||
size_t override_count;
|
||||
};
|
||||
|
||||
#include "codepage_tables.inc"
|
||||
|
||||
constexpr size_t kTableCount = sizeof(kTables) / sizeof(kTables[0]);
|
||||
|
||||
const CodePageTable* find(unsigned cp)
|
||||
{
|
||||
for (const CodePageTable& t : kTables)
|
||||
if (t.cp == cp) return &t;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool lead(const CodePageTable& t, unsigned char b)
|
||||
{
|
||||
for (size_t i = 0; i < t.lead_count; ++i)
|
||||
if (b >= t.leads[i * 2] && b <= t.leads[i * 2 + 1]) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
uint16_t pair(const CodePageTable& t, unsigned char l, unsigned char trail)
|
||||
{
|
||||
for (size_t i = 0; i < t.row_count; ++i) {
|
||||
const DbcsRow& r = t.rows[i];
|
||||
if (r.lead != l) continue;
|
||||
if (trail < r.trail_lo || trail > r.trail_hi) return kNone;
|
||||
return t.dbcs[r.offset + (trail - r.trail_lo)];
|
||||
}
|
||||
return kNone;
|
||||
}
|
||||
|
||||
// Exact encoder (BMP code point -> bytes, kNone when unmapped), built on first use per code page.
|
||||
struct Encoder {
|
||||
std::once_flag once;
|
||||
std::unique_ptr<uint16_t[]> exact;
|
||||
};
|
||||
Encoder g_encoders[kTableCount];
|
||||
|
||||
const uint16_t* exact_encoder(const CodePageTable& t)
|
||||
{
|
||||
Encoder& e = g_encoders[&t - kTables];
|
||||
std::call_once(e.once, [&] {
|
||||
e.exact.reset(new uint16_t[0x10000]);
|
||||
std::fill_n(e.exact.get(), 0x10000, kNone);
|
||||
uint16_t* map = e.exact.get();
|
||||
// First match in ascending byte order, as the generator assumed.
|
||||
for (unsigned b = 0; b < 256; ++b)
|
||||
if (t.single[b] != kNone && map[t.single[b]] == kNone) map[t.single[b]] = uint16_t(b);
|
||||
for (size_t i = 0; i < t.row_count; ++i) {
|
||||
const DbcsRow& r = t.rows[i];
|
||||
for (unsigned trail = r.trail_lo; trail <= r.trail_hi; ++trail) {
|
||||
const uint16_t wc = t.dbcs[r.offset + (trail - r.trail_lo)];
|
||||
if (wc != kNone && map[wc] == kNone) map[wc] = uint16_t(r.lead << 8 | trail);
|
||||
}
|
||||
}
|
||||
for (size_t i = 0; i < t.override_count; ++i) {
|
||||
const EncodeOverride& o = t.overrides[i];
|
||||
if (o.kind == 0) map[o.wc] = o.mb;
|
||||
else if (o.kind == 2) map[o.wc] = kNone;
|
||||
}
|
||||
});
|
||||
return e.exact.get();
|
||||
}
|
||||
|
||||
uint16_t best_fit(const CodePageTable& t, char32_t c)
|
||||
{
|
||||
const EncodeOverride* end = t.overrides + t.override_count;
|
||||
const EncodeOverride* it = std::lower_bound(t.overrides, end, c,
|
||||
[](const EncodeOverride& o, char32_t v) { return o.wc < v; });
|
||||
return it != end && it->wc == c && it->kind == 1 ? it->mb : kNone;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
bool supported(unsigned cp)
|
||||
{
|
||||
return find(cp) != nullptr;
|
||||
}
|
||||
|
||||
bool is_lead_byte(unsigned cp, unsigned char b)
|
||||
{
|
||||
const CodePageTable* t = find(cp);
|
||||
return t && lead(*t, b);
|
||||
}
|
||||
|
||||
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
const CodePageTable* t = find(cp);
|
||||
if (!t) return false;
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < n;) {
|
||||
const unsigned char b = p[i];
|
||||
uint16_t wc = kNone;
|
||||
if (lead(*t, b) && i + 1 < n && p[i + 1] != 0) {
|
||||
wc = pair(*t, b, p[i + 1]);
|
||||
i += 2;
|
||||
} else {
|
||||
wc = t->single[b];
|
||||
++i;
|
||||
}
|
||||
if (wc == kNone) {
|
||||
wc = t->mb_default;
|
||||
ok = false;
|
||||
}
|
||||
out.push_back(wc);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
void encode(unsigned cp, std::u32string_view in, std::string& out, bool allow_best_fit, bool* used_default)
|
||||
{
|
||||
const CodePageTable* t = find(cp);
|
||||
const uint16_t* exact = t ? exact_encoder(*t) : nullptr;
|
||||
for (char32_t c : in) {
|
||||
uint16_t mb = kNone;
|
||||
if (exact && c < 0x10000) {
|
||||
mb = exact[c];
|
||||
if (mb == kNone && allow_best_fit) mb = best_fit(*t, c);
|
||||
}
|
||||
if (mb == kNone) {
|
||||
out.push_back('?');
|
||||
if (used_default) *used_default = true;
|
||||
} else if (mb > 0xFF) {
|
||||
out.push_back(char(mb >> 8));
|
||||
out.push_back(char(mb & 0xFF));
|
||||
} else {
|
||||
out.push_back(char(mb));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mt_codepage
|
||||
@@ -0,0 +1,29 @@
|
||||
#pragma once
|
||||
// Windows ANSI code pages without the host iconv: MultiByteToWideChar / WideCharToMultiByte semantics
|
||||
// driven by Microsoft's WindowsBestFit tables (script/gen_codepage_tables.py). One implementation for
|
||||
// macOS, Linux, Android and iOS, so every platform decodes the 40250 locale and wire text the same way.
|
||||
//
|
||||
// Covered: 874, 932, 936, 949, 950, 1250-1258 (every code page 40250 EterLib/Util.cpp names).
|
||||
// UTF-8 is not handled here.
|
||||
|
||||
#include <cstddef>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
|
||||
namespace mt_codepage {
|
||||
|
||||
bool supported(unsigned cp);
|
||||
|
||||
// Byte string -> code points, as MultiByteToWideChar. A lead byte takes the next non-NUL byte as its
|
||||
// trail; an unmapped byte or pair becomes the code page's default char (U+30FB for 932, else '?').
|
||||
// Returns false when a default char was substituted (MB_ERR_INVALID_CHARS would fail).
|
||||
bool decode(unsigned cp, const unsigned char* p, size_t n, std::u32string& out);
|
||||
|
||||
// Code points -> bytes, as WideCharToMultiByte. With best_fit (no WC_NO_BEST_FIT_CHARS) a character
|
||||
// without an exact mapping may take its WCTABLE best fit ("Ā" -> "A" on 1252); otherwise, and for
|
||||
// anything unmapped, '?' is written and *used_default set.
|
||||
void encode(unsigned cp, std::u32string_view in, std::string& out, bool best_fit, bool* used_default);
|
||||
|
||||
bool is_lead_byte(unsigned cp, unsigned char b);
|
||||
|
||||
} // namespace mt_codepage
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,16 +1,12 @@
|
||||
#include "text_codec.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cerrno>
|
||||
#include <cstring>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include <windows.h>
|
||||
#elif defined(__has_include)
|
||||
#if __has_include(<iconv.h>)
|
||||
#include <iconv.h>
|
||||
#define MT_TEXT_CODEC_HAS_ICONV 1
|
||||
#endif
|
||||
#else
|
||||
// CP936 from Microsoft's own table, not the host iconv (bionic has no GBK).
|
||||
#include "codepage.h"
|
||||
#endif
|
||||
|
||||
namespace mtnet {
|
||||
@@ -69,61 +65,66 @@ bool convert_from_windows(std::string_view input, std::string &out) {
|
||||
return WideCharToMultiByte(CP_UTF8, 0, wide.data(), wide_len, out.data(), utf8_len,
|
||||
nullptr, nullptr) == utf8_len;
|
||||
}
|
||||
#elif defined(MT_TEXT_CODEC_HAS_ICONV)
|
||||
bool convert_iconv(std::string_view input, const char *from, const char *to, std::string &out) {
|
||||
iconv_t cd = iconv_open(to, from);
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
std::string buffer(std::max<size_t>(input.size() * 4 + 16, 32), '\0');
|
||||
char *in_ptr = const_cast<char *>(input.data());
|
||||
size_t in_left = input.size();
|
||||
char *out_ptr = buffer.data();
|
||||
size_t out_left = buffer.size();
|
||||
while (in_left != 0) {
|
||||
const size_t rc = iconv(cd, &in_ptr, &in_left, &out_ptr, &out_left);
|
||||
if (rc == static_cast<size_t>(-1)) {
|
||||
iconv_close(cd);
|
||||
#else
|
||||
// Mirrors the Windows branch: CP936 without best-fit substitutions, and a strict decode.
|
||||
constexpr unsigned kWireCodePage = 936;
|
||||
|
||||
bool decode_utf8_strict(std::string_view input, std::u32string &out) {
|
||||
for (size_t i = 0; i < input.size();) {
|
||||
const size_t seq = utf8_sequence_length(input.data() + i, input.size() - i);
|
||||
if (seq == 0) return false;
|
||||
const unsigned char c = static_cast<unsigned char>(input[i]);
|
||||
char32_t cp = seq == 1 ? c : seq == 2 ? (c & 0x1F) : seq == 3 ? (c & 0x0F) : (c & 0x07);
|
||||
for (size_t k = 1; k < seq; ++k) {
|
||||
const unsigned char t = static_cast<unsigned char>(input[i + k]);
|
||||
if ((t & 0xC0) != 0x80) return false;
|
||||
cp = (cp << 6) | (t & 0x3F);
|
||||
}
|
||||
if ((seq == 3 && cp < 0x800) || (seq == 4 && (cp < 0x10000 || cp > 0x10FFFF)) ||
|
||||
(cp >= 0xD800 && cp <= 0xDFFF)) {
|
||||
return false;
|
||||
}
|
||||
out.push_back(cp);
|
||||
i += seq;
|
||||
}
|
||||
// Flush stateful encodings, even though GB2312 itself is stateless.
|
||||
iconv(cd, nullptr, nullptr, &out_ptr, &out_left);
|
||||
iconv_close(cd);
|
||||
out.assign(buffer.data(), static_cast<size_t>(out_ptr - buffer.data()));
|
||||
return true;
|
||||
}
|
||||
|
||||
void append_utf8(char32_t cp, std::string &out) {
|
||||
if (cp < 0x80) {
|
||||
out.push_back(static_cast<char>(cp));
|
||||
} else if (cp < 0x800) {
|
||||
out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
|
||||
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
||||
} else if (cp < 0x10000) {
|
||||
out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
|
||||
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
|
||||
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
||||
} else {
|
||||
out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
|
||||
out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
|
||||
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
|
||||
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
||||
}
|
||||
}
|
||||
|
||||
bool encode_one(std::string_view input, std::string &out) {
|
||||
return convert_iconv(input, "UTF-8", "GB2312", out) ||
|
||||
convert_iconv(input, "UTF-8", "GBK", out);
|
||||
std::u32string wide;
|
||||
if (!decode_utf8_strict(input, wide)) return false;
|
||||
bool used_default = false;
|
||||
out.clear();
|
||||
mt_codepage::encode(kWireCodePage, wide, out, false, &used_default);
|
||||
return !used_default;
|
||||
}
|
||||
|
||||
bool decode_wire(std::string_view input, std::string &out) {
|
||||
return convert_iconv(input, "GB2312", "UTF-8", out) ||
|
||||
convert_iconv(input, "GBK", "UTF-8", out);
|
||||
}
|
||||
#else
|
||||
bool encode_one(std::string_view input, std::string &out) {
|
||||
// A platform without either Windows code-page APIs or iconv must never
|
||||
// send UTF-8 bytes into a 40250 fixed legacy field. ASCII is lossless;
|
||||
// non-ASCII is reported as unrepresentable so wire_text_fits() rejects it
|
||||
// and to_wire() can safely substitute '?' rather than leaking a 3-byte
|
||||
// UTF-8 sequence into a 24-byte name field.
|
||||
const size_t n = utf8_byte_truncate(input.data(), input.size(), input.size());
|
||||
if (n != input.size()) return false;
|
||||
for (unsigned char c : input) {
|
||||
if (c >= 0x80) return false;
|
||||
std::u32string wide;
|
||||
if (!mt_codepage::decode(kWireCodePage, reinterpret_cast<const unsigned char *>(input.data()),
|
||||
input.size(), wide)) {
|
||||
return false;
|
||||
}
|
||||
out.assign(input.data(), input.size());
|
||||
return true;
|
||||
}
|
||||
|
||||
bool decode_wire(std::string_view input, std::string &out) {
|
||||
// The fallback decoder is deliberately conservative for the same reason:
|
||||
// returning raw high-bit bytes would create invalid UTF-8 in the client.
|
||||
for (unsigned char c : input) {
|
||||
if (c >= 0x80) return false;
|
||||
}
|
||||
out.assign(input.data(), input.size());
|
||||
out.clear();
|
||||
for (char32_t c : wide) append_utf8(c, out);
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
@@ -176,7 +177,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
|
||||
size_t n = 0;
|
||||
while (n < cap && bytes[n] != '\0') ++n;
|
||||
std::string decoded;
|
||||
#if defined(_WIN32) || defined(MT_TEXT_CODEC_HAS_ICONV)
|
||||
// A fixed field can arrive without a terminator and a caller may provide a
|
||||
// larger view than the actual field. If that view ends on a legacy lead byte,
|
||||
// drop only the malformed suffix and preserve the complete preceding text.
|
||||
@@ -188,11 +188,6 @@ std::string from_wire_str(const char *bytes, size_t cap) {
|
||||
#endif
|
||||
if (ok) return decoded;
|
||||
}
|
||||
#else
|
||||
if (decode_wire(std::string_view(bytes, n), decoded)) {
|
||||
return decoded;
|
||||
}
|
||||
#endif
|
||||
// Keep malformed input visible without leaking legacy bytes as invalid UTF-8.
|
||||
decoded.clear();
|
||||
for (size_t i = 0; i < n; ++i) {
|
||||
|
||||
@@ -31,12 +31,12 @@ target_include_directories(port_logic PUBLIC ${MT_PORT_SHIMS})
|
||||
# Libraries 40250 links that are vendored as-is: <lzo/lzo1x.h> (EterBase/lzo.h), <cryptopp/*>
|
||||
# (EterBase/cipher.h), <Python-2.7/*> (ScriptLib).
|
||||
target_link_libraries(port_logic PUBLIC mt3p::minilzo mt3p::cryptopp)
|
||||
# common/Win32Crt.cpp converts non-UTF-8 code pages with iconv. macOS ships it as a separate library;
|
||||
# extension/CMakeLists.txt finds it for mtnet, but the port-only build (script/port_gate.sh) never reads that file.
|
||||
find_library(MT_ICONV_LIBRARY NAMES iconv)
|
||||
if(MT_ICONV_LIBRARY)
|
||||
target_link_libraries(port_logic PUBLIC ${MT_ICONV_LIBRARY})
|
||||
# common/Win32Crt.cpp converts ANSI code pages with mt_codepage. extension/CMakeLists.txt adds it for
|
||||
# mtnet; the port-only build (script/port_gate.sh, native_render) never reads that file.
|
||||
if(NOT TARGET mt_codepage)
|
||||
add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/../codepage ${CMAKE_CURRENT_BINARY_DIR}/../codepage)
|
||||
endif()
|
||||
target_link_libraries(port_logic PUBLIC mt_codepage)
|
||||
if(TARGET mtpython)
|
||||
target_link_libraries(port_logic PUBLIC mt3p::python)
|
||||
endif()
|
||||
@@ -173,6 +173,14 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
|
||||
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})
|
||||
set_tests_properties(port.eterpack PROPERTIES SKIP_RETURN_CODE 77)
|
||||
|
||||
# mt_codepage against Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py caches them
|
||||
# in build/bestfit); skipped (77) without them unless MT_ASSETS_STRICT=1.
|
||||
add_executable(port_codepage_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/codepage_test.cpp)
|
||||
target_link_libraries(port_codepage_test PRIVATE mt_codepage)
|
||||
add_test(NAME port.codepage COMMAND $<TARGET_FILE:port_codepage_test>
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../../../build/bestfit)
|
||||
set_tests_properties(port.codepage PROPERTIES SKIP_RETURN_CODE 77)
|
||||
|
||||
# 批次 2V0-g: the 40250 font path over the FreeType GDI emulation; needs a system Tahoma.
|
||||
add_executable(port_text_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/port_text_test.cpp)
|
||||
target_link_libraries(port_text_test PRIVATE port_platform)
|
||||
|
||||
@@ -11,11 +11,11 @@
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
#if __has_include(<iconv.h>)
|
||||
#include <iconv.h>
|
||||
#define MT_WIN32CRT_HAS_ICONV 1
|
||||
#endif
|
||||
#include <glob.h>
|
||||
// ANSI code pages come from Microsoft's own tables, not the host iconv (bionic has no GBK).
|
||||
#include "codepage.h"
|
||||
#include <algorithm>
|
||||
#include <dirent.h>
|
||||
#include <fnmatch.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
@@ -103,17 +103,30 @@ HANDLE FindFirstFile(const char* pattern, WIN32_FIND_DATA* data)
|
||||
std::string normalized(pattern);
|
||||
for (char& ch : normalized)
|
||||
if (ch == '\\') ch = '/';
|
||||
glob_t matches{};
|
||||
const int result = glob(normalized.c_str(), 0, nullptr, &matches);
|
||||
if (result != 0 || matches.gl_pathc == 0)
|
||||
// Win32 wildcards only appear in the last component; match it against the directory like
|
||||
// glob(3) would (sorted, '*' skips dot files). glob itself is API 28+ on Android.
|
||||
std::vector<std::string> paths;
|
||||
const size_t slash = normalized.rfind('/');
|
||||
const std::string dir = slash == std::string::npos ? std::string() : normalized.substr(0, slash + 1);
|
||||
const std::string name = normalized.substr(dir.size());
|
||||
if (name.find_first_of("*?[") == std::string::npos)
|
||||
{
|
||||
globfree(&matches);
|
||||
return INVALID_HANDLE_VALUE;
|
||||
struct stat st;
|
||||
if (::stat(normalized.c_str(), &st) == 0)
|
||||
paths.push_back(normalized);
|
||||
}
|
||||
else if (DIR* d = ::opendir(dir.empty() ? "." : dir.c_str()))
|
||||
{
|
||||
while (const dirent* entry = ::readdir(d))
|
||||
if (::fnmatch(name.c_str(), entry->d_name, FNM_PERIOD) == 0)
|
||||
paths.push_back(dir + entry->d_name);
|
||||
::closedir(d);
|
||||
std::sort(paths.begin(), paths.end());
|
||||
}
|
||||
if (paths.empty())
|
||||
return INVALID_HANDLE_VALUE;
|
||||
auto* state = new FindState;
|
||||
for (size_t i = 0; i < matches.gl_pathc; ++i)
|
||||
state->paths.emplace_back(matches.gl_pathv[i]);
|
||||
globfree(&matches);
|
||||
state->paths = std::move(paths);
|
||||
fill_find_data(state->paths[0], data);
|
||||
state->next = 1;
|
||||
return state;
|
||||
@@ -199,13 +212,8 @@ DWORD GetTickCount()
|
||||
|
||||
namespace {
|
||||
|
||||
// Windows-1252 0x80..0x9F; the other bytes map to the same code point (0 marks the undefined ones).
|
||||
const char32_t kCp1252High[32] = {
|
||||
0x20AC, 0, 0x201A, 0x0192, 0x201E, 0x2026, 0x2020, 0x2021, 0x02C6, 0x2030, 0x0160, 0x2039, 0x0152, 0, 0x017D, 0,
|
||||
0, 0x2018, 0x2019, 0x201C, 0x201D, 0x2022, 0x2013, 0x2014, 0x02DC, 0x2122, 0x0161, 0x203A, 0x0153, 0, 0x017E, 0x0178,
|
||||
};
|
||||
|
||||
bool is_cp1252(UINT cp) { return cp == CP_ACP || cp == 1252; }
|
||||
// CP_ACP is the Western system code page here, as on the 40250 target machines.
|
||||
UINT ansi(UINT cp) { return cp == CP_ACP ? 1252 : cp; }
|
||||
|
||||
bool decode_utf8(const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
@@ -237,108 +245,25 @@ void encode_utf8(char32_t cp, std::string& out)
|
||||
else { out.push_back(char(0xF0 | (cp >> 18))); out.push_back(char(0x80 | ((cp >> 12) & 0x3F))); out.push_back(char(0x80 | ((cp >> 6) & 0x3F))); out.push_back(char(0x80 | (cp & 0x3F))); }
|
||||
}
|
||||
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
std::string iconv_name(UINT cp)
|
||||
{
|
||||
switch (cp) {
|
||||
case 949: return "CP949";
|
||||
case 936: return "GBK";
|
||||
case 950: return "BIG5";
|
||||
case 932: return "SHIFT_JIS";
|
||||
case 874: return "CP874";
|
||||
default: return "CP" + std::to_string(cp);
|
||||
}
|
||||
}
|
||||
|
||||
// One character at a time, so an undecodable byte becomes one default char like Windows does.
|
||||
bool iconv_decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
iconv_t cd = iconv_open("UTF-32LE", iconv_name(cp).c_str());
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
for (size_t i = 0; i < n;) {
|
||||
unsigned char wide[16];
|
||||
bool done = false;
|
||||
for (size_t len = 1; len <= 4 && i + len <= n && !done; ++len) {
|
||||
iconv(cd, nullptr, nullptr, nullptr, nullptr);
|
||||
char* in = const_cast<char*>(reinterpret_cast<const char*>(p + i));
|
||||
size_t in_left = len;
|
||||
char* o = reinterpret_cast<char*>(wide);
|
||||
size_t o_left = sizeof(wide);
|
||||
if (iconv(cd, &in, &in_left, &o, &o_left) != size_t(-1) && in_left == 0 && o_left < sizeof(wide)) {
|
||||
for (size_t k = 0; k + 4 <= sizeof(wide) - o_left; k += 4)
|
||||
out.push_back(char32_t(wide[k] | (wide[k + 1] << 8) | (wide[k + 2] << 16) | (unsigned(wide[k + 3]) << 24)));
|
||||
i += len;
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
if (!done) { out.push_back(0xFFFD); ++i; }
|
||||
}
|
||||
iconv_close(cd);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool iconv_encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
|
||||
{
|
||||
iconv_t cd = iconv_open(iconv_name(cp).c_str(), "UTF-32LE");
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
for (char32_t c : in) {
|
||||
unsigned char src[4] = {static_cast<unsigned char>(c), static_cast<unsigned char>(c >> 8), static_cast<unsigned char>(c >> 16), static_cast<unsigned char>(c >> 24)};
|
||||
char buf[16];
|
||||
char* ip = reinterpret_cast<char*>(src);
|
||||
size_t il = 4;
|
||||
char* op = buf;
|
||||
size_t ol = sizeof(buf);
|
||||
iconv(cd, nullptr, nullptr, nullptr, nullptr);
|
||||
if (iconv(cd, &ip, &il, &op, &ol) != size_t(-1) && il == 0) {
|
||||
out.append(buf, sizeof(buf) - ol);
|
||||
} else {
|
||||
out.push_back('?');
|
||||
*used_default = true;
|
||||
}
|
||||
}
|
||||
iconv_close(cd);
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
bool decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
if (cp == CP_UTF8) return decode_utf8(p, n, out);
|
||||
if (is_cp1252(cp)) {
|
||||
for (size_t i = 0; i < n; ++i) {
|
||||
const unsigned char c = p[i];
|
||||
out.push_back(c >= 0x80 && c < 0xA0 ? (kCp1252High[c - 0x80] ? kCp1252High[c - 0x80] : char32_t(c)) : char32_t(c));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
if (iconv_decode(cp, p, n, out)) return true;
|
||||
#endif
|
||||
// No converter for this code page: ASCII passes, the rest becomes U+FFFD.
|
||||
if (mt_codepage::supported(ansi(cp))) return mt_codepage::decode(ansi(cp), p, n, out);
|
||||
// No table for this code page: ASCII passes, the rest becomes U+FFFD.
|
||||
for (size_t i = 0; i < n; ++i) out.push_back(p[i] < 0x80 ? char32_t(p[i]) : char32_t(0xFFFD));
|
||||
return true;
|
||||
}
|
||||
|
||||
void encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
|
||||
void encode(UINT cp, DWORD flags, const std::u32string& in, std::string& out, bool* used_default)
|
||||
{
|
||||
if (cp == CP_UTF8) {
|
||||
for (char32_t c : in) encode_utf8(c, out);
|
||||
return;
|
||||
}
|
||||
if (is_cp1252(cp)) {
|
||||
for (char32_t c : in) {
|
||||
if (c < 0x80 || (c >= 0xA0 && c < 0x100)) { out.push_back(char(c)); continue; }
|
||||
int hit = -1;
|
||||
for (int k = 0; k < 32; ++k)
|
||||
if (kCp1252High[k] == c) { hit = k; break; }
|
||||
if (hit >= 0) out.push_back(char(0x80 + hit));
|
||||
else { out.push_back('?'); *used_default = true; }
|
||||
}
|
||||
if (mt_codepage::supported(ansi(cp))) {
|
||||
mt_codepage::encode(ansi(cp), in, out, (flags & WC_NO_BEST_FIT_CHARS) == 0, used_default);
|
||||
return;
|
||||
}
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
if (iconv_encode(cp, in, out, used_default)) return;
|
||||
#endif
|
||||
for (char32_t c : in) {
|
||||
if (c < 0x80) out.push_back(char(c));
|
||||
else { out.push_back('?'); *used_default = true; }
|
||||
@@ -361,7 +286,7 @@ int MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int
|
||||
return int(wide.size());
|
||||
}
|
||||
|
||||
int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWideChar,
|
||||
int WideCharToMultiByte(UINT CodePage, DWORD dwFlags, LPCWSTR lpWideCharStr, int cchWideChar,
|
||||
LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar, LPBOOL lpUsedDefaultChar)
|
||||
{
|
||||
if (!lpWideCharStr || cchWideChar == 0 || cbMultiByte < 0) return 0;
|
||||
@@ -372,14 +297,14 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
|
||||
for (size_t i = 0; i < n; ++i) wide[i] = char32_t(lpWideCharStr[i]);
|
||||
std::string bytes;
|
||||
bool used_default = false;
|
||||
encode(CodePage, wide, bytes, &used_default);
|
||||
encode(CodePage, dwFlags, wide, bytes, &used_default);
|
||||
if (used_default && lpDefaultChar && CodePage != CP_UTF8) {
|
||||
// '?' was the placeholder; the caller's default char replaces it only where a fallback happened.
|
||||
std::string redo;
|
||||
for (char32_t c : wide) {
|
||||
std::string one;
|
||||
bool miss = false;
|
||||
encode(CodePage, std::u32string(1, c), one, &miss);
|
||||
encode(CodePage, dwFlags, std::u32string(1, c), one, &miss);
|
||||
redo += miss ? std::string(1, *lpDefaultChar) : one;
|
||||
}
|
||||
bytes.swap(redo);
|
||||
@@ -393,13 +318,7 @@ int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWide
|
||||
|
||||
BOOL IsDBCSLeadByteEx(UINT CodePage, BYTE TestChar)
|
||||
{
|
||||
switch (CodePage) {
|
||||
case 932: return (TestChar >= 0x81 && TestChar <= 0x9F) || (TestChar >= 0xE0 && TestChar <= 0xFC);
|
||||
case 936:
|
||||
case 949:
|
||||
case 950: return TestChar >= 0x81 && TestChar <= 0xFE;
|
||||
default: return FALSE;
|
||||
}
|
||||
return mt_codepage::is_lead_byte(ansi(CodePage), TestChar) ? TRUE : FALSE;
|
||||
}
|
||||
|
||||
LPSTR CharNextExA(WORD CodePage, LPCSTR lpCurrentChar, DWORD)
|
||||
|
||||
@@ -104,8 +104,8 @@ void DeleteCriticalSection(LPCRITICAL_SECTION cs);
|
||||
void EnterCriticalSection(LPCRITICAL_SECTION cs);
|
||||
void LeaveCriticalSection(LPCRITICAL_SECTION cs);
|
||||
|
||||
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 and
|
||||
// CP_ACP/1252 are converted directly; other code pages go through iconv where the platform has it.
|
||||
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 is converted
|
||||
// directly; CP_ACP (= 1252) and the other ANSI code pages use Microsoft's tables (codepage/codepage.h).
|
||||
#define CP_ACP 0
|
||||
#define CP_UTF8 65001
|
||||
#define MB_PRECOMPOSED 0x00000001
|
||||
|
||||
Reference in New Issue
Block a user