mt_codepage implements code pages 874/932/936/949/950/1250-1258 from Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py, SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/ IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build no longer depends on iconv. port.codepage checks every table against its source file. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
215 lines
8.5 KiB
C++
215 lines
8.5 KiB
C++
// codepage_test — mt_codepage against Microsoft's WindowsBestFit source tables, plus spot checks of the
|
|
// characters each shipped 40250 locale needs.
|
|
//
|
|
// Argument: the directory holding bestfit<cp>.txt (script/gen_codepage_tables.py downloads them to
|
|
// build/bestfit). Without it only the spot checks run and the test exits 77 (ctest SKIP), unless
|
|
// MT_ASSETS_STRICT=1.
|
|
#include "codepage.h"
|
|
|
|
#include <cstdio>
|
|
#include <cstdlib>
|
|
#include <cstring>
|
|
#include <fstream>
|
|
#include <map>
|
|
#include <sstream>
|
|
#include <string>
|
|
#include <vector>
|
|
|
|
static int g_failures = 0;
|
|
#define CHECK(cond) \
|
|
do { \
|
|
if (!(cond)) { \
|
|
std::fprintf(stderr, "%s:%d: CHECK(%s)\n", __FILE__, __LINE__, #cond); \
|
|
++g_failures; \
|
|
} \
|
|
} while (0)
|
|
|
|
static std::u32string dec(unsigned cp, const char* bytes, bool* ok = nullptr)
|
|
{
|
|
std::u32string out;
|
|
const bool r = mt_codepage::decode(cp, reinterpret_cast<const unsigned char*>(bytes), std::strlen(bytes), out);
|
|
if (ok) *ok = r;
|
|
return out;
|
|
}
|
|
|
|
static std::string enc(unsigned cp, std::u32string in, bool best_fit = false, bool* used_default = nullptr)
|
|
{
|
|
std::string out;
|
|
bool used = false;
|
|
mt_codepage::encode(cp, in, out, best_fit, &used);
|
|
if (used_default) *used_default = used;
|
|
return out;
|
|
}
|
|
|
|
static void spot_checks()
|
|
{
|
|
// zh (936): "中文" D6D0 CEC4, and the Windows-only single byte 0x80 = euro.
|
|
CHECK(dec(936, "\xd6\xd0\xce\xc4") == U"中文");
|
|
CHECK(enc(936, U"中文") == "\xd6\xd0\xce\xc4");
|
|
CHECK(dec(936, "\x80") == U"€");
|
|
// GBK extension beyond GB2312 (U+4E02 = 81 40), both directions.
|
|
CHECK(dec(936, "\x81\x40") == U"丂");
|
|
CHECK(enc(936, U"丂") == "\x81\x40");
|
|
// A lead byte with nothing after it is one default char and reports failure.
|
|
bool ok = true;
|
|
CHECK(dec(936, "a\xd6", &ok) == U"a?" && !ok);
|
|
// An unmapped pair is consumed whole (932 uses U+30FB as its default).
|
|
CHECK(dec(932, "\x85\x40") == U"・");
|
|
CHECK(dec(932, "\x82\xa0") == U"あ");
|
|
|
|
// The shipped European locales.
|
|
CHECK(dec(1252, "\x80\xe9") == U"€é"); // en/de/fr/...: euro, é
|
|
CHECK(dec(1252, "\x81") == U"\u0081"); // undefined byte keeps its value, as Windows
|
|
CHECK(dec(1250, "\xb9\xb3\x9c") == U"ąłś"); // pl/cz/hu/ro: ą ł ś
|
|
CHECK(dec(1251, "\xc6\xe8") == U"Жи"); // ru: Жи
|
|
CHECK(dec(1253, "\xd9\xe1") == U"Ωα"); // gr: Ωα
|
|
CHECK(dec(1254, "\xf0\xfe\xdd") == U"ğşİ"); // tr: ğ ş İ
|
|
CHECK(enc(1250, U"ąłś") == "\xb9\xb3\x9c");
|
|
CHECK(enc(1251, U"Жи") == "\xc6\xe8");
|
|
CHECK(enc(1253, U"Ωα") == "\xd9\xe1");
|
|
CHECK(enc(1254, U"ğşİ") == "\xf0\xfe\xdd");
|
|
|
|
// Best fit only when WC_NO_BEST_FIT_CHARS is absent: 1252 has no "Ā", its best fit is "A".
|
|
bool used = false;
|
|
CHECK(enc(1252, U"Ā", true, &used) == "A" && !used);
|
|
CHECK(enc(1252, U"Ā", false, &used) == "?" && used);
|
|
// Unmapped either way; beyond the BMP too.
|
|
CHECK(enc(1251, U"中", true, &used) == "?" && used);
|
|
CHECK(enc(936, U"\U0001F600", true, &used) == "?" && used);
|
|
|
|
CHECK(mt_codepage::is_lead_byte(936, 0x81) && !mt_codepage::is_lead_byte(936, 0x80));
|
|
CHECK(mt_codepage::is_lead_byte(932, 0xe0) && !mt_codepage::is_lead_byte(932, 0xa1));
|
|
CHECK(!mt_codepage::is_lead_byte(1250, 0xb9));
|
|
CHECK(!mt_codepage::supported(65001) && !mt_codepage::supported(437));
|
|
}
|
|
|
|
struct SourceTable {
|
|
unsigned mb_default = 0x3f;
|
|
std::map<unsigned, unsigned> decode; // bytes (pair = lead << 8 | trail) -> wc
|
|
std::map<unsigned, unsigned> wctable;
|
|
std::vector<std::pair<unsigned, unsigned>> leads;
|
|
};
|
|
|
|
static bool load(const std::string& path, SourceTable& t)
|
|
{
|
|
std::ifstream in(path, std::ios::binary);
|
|
if (!in) return false;
|
|
std::string raw, section;
|
|
unsigned lead = 0;
|
|
while (std::getline(in, raw)) {
|
|
std::string line = raw.substr(0, raw.find(';'));
|
|
std::istringstream words(line);
|
|
std::string head;
|
|
if (!(words >> head)) continue;
|
|
if (head == "CPINFO") {
|
|
std::string size, byte_default, wc_default;
|
|
words >> size >> byte_default >> wc_default;
|
|
t.mb_default = unsigned(std::stoul(wc_default, nullptr, 16));
|
|
continue;
|
|
}
|
|
if (head == "MBTABLE" || head == "WCTABLE" || head == "DBCSRANGE") { section = head; continue; }
|
|
if (head == "DBCSTABLE") {
|
|
section = head;
|
|
lead = unsigned(std::stoul(raw.substr(raw.find("0x", raw.find("LeadByte"))), nullptr, 16));
|
|
continue;
|
|
}
|
|
if (head == "ENDCODEPAGE") break;
|
|
if (head.rfind("0x", 0) != 0) continue;
|
|
std::string second;
|
|
words >> second;
|
|
const unsigned a = unsigned(std::stoul(head, nullptr, 16));
|
|
const unsigned b = unsigned(std::stoul(second, nullptr, 16));
|
|
if (raw.find("Lead Byte Range") != std::string::npos) t.leads.push_back({a, b});
|
|
else if (section == "MBTABLE") t.decode[a] = b;
|
|
else if (section == "DBCSTABLE") t.decode[lead << 8 | a] = b;
|
|
else if (section == "WCTABLE") t.wctable[a] = b;
|
|
}
|
|
return !t.decode.empty() && !t.wctable.empty();
|
|
}
|
|
|
|
static std::string bytes_of(unsigned mb)
|
|
{
|
|
return mb > 0xff ? std::string{char(mb >> 8), char(mb & 0xff)} : std::string(1, char(mb));
|
|
}
|
|
|
|
// Every byte / pair the source defines decodes to it, every other one to the default char; every
|
|
// WCTABLE entry encodes to it with best fit, and without best fit only when it round-trips.
|
|
static int against_source(const std::string& dir)
|
|
{
|
|
const unsigned cps[] = {874, 932, 936, 949, 950, 1250, 1251, 1252, 1253, 1254, 1255, 1256, 1257, 1258};
|
|
int loaded = 0;
|
|
for (unsigned cp : cps) {
|
|
SourceTable t;
|
|
if (!load(dir + "/bestfit" + std::to_string(cp) + ".txt", t)) continue;
|
|
++loaded;
|
|
const int before = g_failures;
|
|
auto is_lead = [&](unsigned b) {
|
|
for (auto& r : t.leads) if (b >= r.first && b <= r.second) return true;
|
|
return false;
|
|
};
|
|
for (unsigned b = 0; b < 256; ++b) {
|
|
CHECK(mt_codepage::is_lead_byte(cp, (unsigned char)b) == is_lead(b));
|
|
if (is_lead(b)) {
|
|
for (unsigned trail = 1; trail < 256; ++trail) {
|
|
const unsigned char pair[2] = {(unsigned char)b, (unsigned char)trail};
|
|
std::u32string out;
|
|
const bool ok = mt_codepage::decode(cp, pair, 2, out);
|
|
auto it = t.decode.find(b << 8 | trail);
|
|
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
|
|
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
|
|
std::fprintf(stderr, "cp%u %02x%02x -> %x, want %x\n", cp, b, trail,
|
|
out.empty() ? 0 : unsigned(out[0]), want);
|
|
++g_failures;
|
|
}
|
|
}
|
|
} else if (b != 0) {
|
|
const unsigned char one[1] = {(unsigned char)b};
|
|
std::u32string out;
|
|
const bool ok = mt_codepage::decode(cp, one, 1, out);
|
|
auto it = t.decode.find(b);
|
|
const unsigned want = it == t.decode.end() ? t.mb_default : it->second;
|
|
if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) {
|
|
std::fprintf(stderr, "cp%u %02x -> %x, want %x\n", cp, b, out.empty() ? 0 : unsigned(out[0]), want);
|
|
++g_failures;
|
|
}
|
|
}
|
|
}
|
|
for (unsigned wc = 1; wc < 0x10000; ++wc) {
|
|
if (wc >= 0xd800 && wc <= 0xdfff) continue;
|
|
auto it = t.wctable.find(wc);
|
|
bool used = false;
|
|
const std::string fit = enc(cp, std::u32string(1, char32_t(wc)), true, &used);
|
|
const std::string want_fit = it == t.wctable.end() ? "?" : bytes_of(it->second);
|
|
const bool exact = it != t.wctable.end() && t.decode.count(it->second) && t.decode[it->second] == wc;
|
|
const std::string strict = enc(cp, std::u32string(1, char32_t(wc)), false);
|
|
const std::string want_strict = exact ? bytes_of(it->second) : "?";
|
|
if (fit != want_fit || used != (it == t.wctable.end()) || strict != want_strict) {
|
|
std::fprintf(stderr, "cp%u U+%04X -> best fit %zu bytes / strict %zu bytes, mismatch\n", cp, wc,
|
|
fit.size(), strict.size());
|
|
++g_failures;
|
|
}
|
|
}
|
|
std::printf("cp%u: %zu byte sequences, %zu WCTABLE entries: %s\n", cp, t.decode.size(), t.wctable.size(),
|
|
g_failures == before ? "ok" : "FAIL");
|
|
}
|
|
return loaded;
|
|
}
|
|
|
|
int main(int argc, char** argv)
|
|
{
|
|
spot_checks();
|
|
const char* strict = std::getenv("MT_ASSETS_STRICT");
|
|
const bool is_strict = strict && std::string(strict) == "1";
|
|
const int loaded = argc > 1 ? against_source(argv[1]) : 0;
|
|
if (g_failures) {
|
|
std::fprintf(stderr, "%d failure(s)\n", g_failures);
|
|
return 1;
|
|
}
|
|
if (loaded < 14) {
|
|
std::printf("bestfit sources: %d of 14 found (run script/gen_codepage_tables.py)\n", loaded);
|
|
return is_strict ? 1 : 77;
|
|
}
|
|
std::printf("ok\n");
|
|
return 0;
|
|
}
|