// codepage_test — mt_codepage against Microsoft's WindowsBestFit source tables, plus spot checks of the // characters each shipped 40250 locale needs. // // Argument: the directory holding bestfit.txt (script/gen_codepage_tables.py downloads them to // build/bestfit). Without it only the spot checks run and the test exits 77 (ctest SKIP), unless // MT_ASSETS_STRICT=1. #include "codepage.h" #include #include #include #include #include #include #include #include static int g_failures = 0; #define CHECK(cond) \ do { \ if (!(cond)) { \ std::fprintf(stderr, "%s:%d: CHECK(%s)\n", __FILE__, __LINE__, #cond); \ ++g_failures; \ } \ } while (0) static std::u32string dec(unsigned cp, const char* bytes, bool* ok = nullptr) { std::u32string out; const bool r = mt_codepage::decode(cp, reinterpret_cast(bytes), std::strlen(bytes), out); if (ok) *ok = r; return out; } static std::string enc(unsigned cp, std::u32string in, bool best_fit = false, bool* used_default = nullptr) { std::string out; bool used = false; mt_codepage::encode(cp, in, out, best_fit, &used); if (used_default) *used_default = used; return out; } static void spot_checks() { // zh (936): "中文" D6D0 CEC4, and the Windows-only single byte 0x80 = euro. CHECK(dec(936, "\xd6\xd0\xce\xc4") == U"中文"); CHECK(enc(936, U"中文") == "\xd6\xd0\xce\xc4"); CHECK(dec(936, "\x80") == U"€"); // GBK extension beyond GB2312 (U+4E02 = 81 40), both directions. CHECK(dec(936, "\x81\x40") == U"丂"); CHECK(enc(936, U"丂") == "\x81\x40"); // A lead byte with nothing after it is one default char and reports failure. bool ok = true; CHECK(dec(936, "a\xd6", &ok) == U"a?" && !ok); // An unmapped pair is consumed whole (932 uses U+30FB as its default). CHECK(dec(932, "\x85\x40") == U"・"); CHECK(dec(932, "\x82\xa0") == U"あ"); // The shipped European locales. CHECK(dec(1252, "\x80\xe9") == U"€é"); // en/de/fr/...: euro, é CHECK(dec(1252, "\x81") == U"\u0081"); // undefined byte keeps its value, as Windows CHECK(dec(1250, "\xb9\xb3\x9c") == U"ąłś"); // pl/cz/hu/ro: ą ł ś CHECK(dec(1251, "\xc6\xe8") == U"Жи"); // ru: Жи CHECK(dec(1253, "\xd9\xe1") == U"Ωα"); // gr: Ωα CHECK(dec(1254, "\xf0\xfe\xdd") == U"ğşİ"); // tr: ğ ş İ CHECK(enc(1250, U"ąłś") == "\xb9\xb3\x9c"); CHECK(enc(1251, U"Жи") == "\xc6\xe8"); CHECK(enc(1253, U"Ωα") == "\xd9\xe1"); CHECK(enc(1254, U"ğşİ") == "\xf0\xfe\xdd"); // Best fit only when WC_NO_BEST_FIT_CHARS is absent: 1252 has no "Ā", its best fit is "A". bool used = false; CHECK(enc(1252, U"Ā", true, &used) == "A" && !used); CHECK(enc(1252, U"Ā", false, &used) == "?" && used); // Unmapped either way; beyond the BMP too. CHECK(enc(1251, U"中", true, &used) == "?" && used); CHECK(enc(936, U"\U0001F600", true, &used) == "?" && used); CHECK(mt_codepage::is_lead_byte(936, 0x81) && !mt_codepage::is_lead_byte(936, 0x80)); CHECK(mt_codepage::is_lead_byte(932, 0xe0) && !mt_codepage::is_lead_byte(932, 0xa1)); CHECK(!mt_codepage::is_lead_byte(1250, 0xb9)); CHECK(!mt_codepage::supported(65001) && !mt_codepage::supported(437)); } struct SourceTable { unsigned mb_default = 0x3f; std::map decode; // bytes (pair = lead << 8 | trail) -> wc std::map wctable; std::vector> leads; }; static bool load(const std::string& path, SourceTable& t) { std::ifstream in(path, std::ios::binary); if (!in) return false; std::string raw, section; unsigned lead = 0; while (std::getline(in, raw)) { std::string line = raw.substr(0, raw.find(';')); std::istringstream words(line); std::string head; if (!(words >> head)) continue; if (head == "CPINFO") { std::string size, byte_default, wc_default; words >> size >> byte_default >> wc_default; t.mb_default = unsigned(std::stoul(wc_default, nullptr, 16)); continue; } if (head == "MBTABLE" || head == "WCTABLE" || head == "DBCSRANGE") { section = head; continue; } if (head == "DBCSTABLE") { section = head; lead = unsigned(std::stoul(raw.substr(raw.find("0x", raw.find("LeadByte"))), nullptr, 16)); continue; } if (head == "ENDCODEPAGE") break; if (head.rfind("0x", 0) != 0) continue; std::string second; words >> second; const unsigned a = unsigned(std::stoul(head, nullptr, 16)); const unsigned b = unsigned(std::stoul(second, nullptr, 16)); if (raw.find("Lead Byte Range") != std::string::npos) t.leads.push_back({a, b}); else if (section == "MBTABLE") t.decode[a] = b; else if (section == "DBCSTABLE") t.decode[lead << 8 | a] = b; else if (section == "WCTABLE") t.wctable[a] = b; } return !t.decode.empty() && !t.wctable.empty(); } static std::string bytes_of(unsigned mb) { return mb > 0xff ? std::string{char(mb >> 8), char(mb & 0xff)} : std::string(1, char(mb)); } // Every byte / pair the source defines decodes to it, every other one to the default char; every // WCTABLE entry encodes to it with best fit, and without best fit only when it round-trips. static int against_source(const std::string& dir) { const unsigned cps[] = {874, 932, 936, 949, 950, 1250, 1251, 1252, 1253, 1254, 1255, 1256, 1257, 1258}; int loaded = 0; for (unsigned cp : cps) { SourceTable t; if (!load(dir + "/bestfit" + std::to_string(cp) + ".txt", t)) continue; ++loaded; const int before = g_failures; auto is_lead = [&](unsigned b) { for (auto& r : t.leads) if (b >= r.first && b <= r.second) return true; return false; }; for (unsigned b = 0; b < 256; ++b) { CHECK(mt_codepage::is_lead_byte(cp, (unsigned char)b) == is_lead(b)); if (is_lead(b)) { for (unsigned trail = 1; trail < 256; ++trail) { const unsigned char pair[2] = {(unsigned char)b, (unsigned char)trail}; std::u32string out; const bool ok = mt_codepage::decode(cp, pair, 2, out); auto it = t.decode.find(b << 8 | trail); const unsigned want = it == t.decode.end() ? t.mb_default : it->second; if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) { std::fprintf(stderr, "cp%u %02x%02x -> %x, want %x\n", cp, b, trail, out.empty() ? 0 : unsigned(out[0]), want); ++g_failures; } } } else if (b != 0) { const unsigned char one[1] = {(unsigned char)b}; std::u32string out; const bool ok = mt_codepage::decode(cp, one, 1, out); auto it = t.decode.find(b); const unsigned want = it == t.decode.end() ? t.mb_default : it->second; if (out.size() != 1 || out[0] != want || ok != (it != t.decode.end())) { std::fprintf(stderr, "cp%u %02x -> %x, want %x\n", cp, b, out.empty() ? 0 : unsigned(out[0]), want); ++g_failures; } } } for (unsigned wc = 1; wc < 0x10000; ++wc) { if (wc >= 0xd800 && wc <= 0xdfff) continue; auto it = t.wctable.find(wc); bool used = false; const std::string fit = enc(cp, std::u32string(1, char32_t(wc)), true, &used); const std::string want_fit = it == t.wctable.end() ? "?" : bytes_of(it->second); const bool exact = it != t.wctable.end() && t.decode.count(it->second) && t.decode[it->second] == wc; const std::string strict = enc(cp, std::u32string(1, char32_t(wc)), false); const std::string want_strict = exact ? bytes_of(it->second) : "?"; if (fit != want_fit || used != (it == t.wctable.end()) || strict != want_strict) { std::fprintf(stderr, "cp%u U+%04X -> best fit %zu bytes / strict %zu bytes, mismatch\n", cp, wc, fit.size(), strict.size()); ++g_failures; } } std::printf("cp%u: %zu byte sequences, %zu WCTABLE entries: %s\n", cp, t.decode.size(), t.wctable.size(), g_failures == before ? "ok" : "FAIL"); } return loaded; } int main(int argc, char** argv) { spot_checks(); const char* strict = std::getenv("MT_ASSETS_STRICT"); const bool is_strict = strict && std::string(strict) == "1"; const int loaded = argc > 1 ? against_source(argv[1]) : 0; if (g_failures) { std::fprintf(stderr, "%d failure(s)\n", g_failures); return 1; } if (loaded < 14) { std::printf("bestfit sources: %d of 14 found (run script/gen_codepage_tables.py)\n", loaded); return is_strict ? 1 : 77; } std::printf("ok\n"); return 0; }