#!/usr/bin/env python3 """Generate extension/src/codepage/codepage_tables.inc from Microsoft's WindowsBestFit tables. The port emulates MultiByteToWideChar / WideCharToMultiByte off Windows. Instead of the host iconv (Android's bionic has no GBK, and iconv tables differ from Windows in the corners), the conversions use Microsoft's own code page definitions, published by unicode.org: https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit.txt Each file has an MBTABLE (single bytes), DBCSRANGE/DBCSTABLE (lead byte + trail byte) and a WCTABLE (Unicode -> bytes, including the many-to-one "best fit" entries WideCharToMultiByte uses without WC_NO_BEST_FIT_CHARS). The generated data keeps the byte -> Unicode side and only the WCTABLE entries that differ from its first-match inverse, so the runtime can rebuild the encoder. Usage: python3 script/gen_codepage_tables.py [--cache DIR] [--output FILE] Downloads missing files into the cache (build/bestfit by default) and checks their SHA-256. """ import argparse import hashlib import re import sys import urllib.request from pathlib import Path REPO = Path(__file__).resolve().parent.parent URL = "https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit{}.txt" # Every code page 40250 EterLib/Util.cpp names (1252 included; CP_ACP maps to it). SHA256 = { 874: "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd", 932: "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331", 936: "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae", 949: "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a", 950: "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a", 1250: "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b", 1251: "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7", 1252: "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9", 1253: "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e", 1254: "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a", 1255: "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a", 1256: "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007", 1257: "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366", 1258: "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1", } NONE = 0xFFFF PAIR = re.compile(rb"^\s*0x([0-9a-fA-F]+)\s+0x([0-9a-fA-F]+)") def fetch(cp: int, cache: Path) -> bytes: path = cache / f"bestfit{cp}.txt" if not path.is_file(): cache.mkdir(parents=True, exist_ok=True) with urllib.request.urlopen(URL.format(cp)) as response: path.write_bytes(response.read()) data = path.read_bytes() digest = hashlib.sha256(data).hexdigest() if digest != SHA256[cp]: raise SystemExit(f"{path}: SHA-256 {digest} != {SHA256[cp]}") return data def parse(cp: int, data: bytes): single = [NONE] * 256 leads = [] # [(lo, hi)] dbcs = {} # lead -> {trail: wc} wctable = {} # wc -> mb mb_default = None section = None lead = None for raw in data.splitlines(): line = raw.split(b";", 1)[0].strip() if not line: continue head = line.split()[0] if head == b"CODEPAGE": if int(line.split()[1]) != cp: raise SystemExit(f"bestfit{cp}.txt declares {line!r}") continue if head == b"CPINFO": fields = line.split() if int(fields[2], 16) != 0x3F: raise SystemExit(f"cp{cp}: byte default char {fields[2]!r} is not '?'") mb_default = int(fields[3], 16) continue if head in (b"MBTABLE", b"WCTABLE", b"DBCSRANGE"): section = head continue if head == b"DBCSTABLE": section = head match = re.search(rb"LeadByte\s*=\s*0x([0-9a-fA-F]+)", raw) lead = int(match.group(1), 16) dbcs.setdefault(lead, {}) continue if head == b"ENDCODEPAGE": break match = PAIR.match(line) if not match: raise SystemExit(f"cp{cp}: unexpected line {raw!r}") a, b = int(match.group(1), 16), int(match.group(2), 16) # cp932 lists its second lead range (0xe0-0xfc) after the first range's DBCSTABLEs. if b"Lead Byte Range" in raw and section in (b"DBCSRANGE", b"DBCSTABLE"): leads.append((a, b)) continue if section == b"MBTABLE": single[a] = b elif section == b"DBCSRANGE": leads.append((a, b)) elif section == b"DBCSTABLE": dbcs[lead][a] = b elif section == b"WCTABLE": wctable[a] = b if mb_default is None: raise SystemExit(f"cp{cp}: no CPINFO") for lead in dbcs: if not any(lo <= lead <= hi for lo, hi in leads): raise SystemExit(f"cp{cp}: DBCSTABLE 0x{lead:02x} outside the lead ranges {leads}") return single, leads, dbcs, wctable, mb_default def decode_map(single, dbcs): """bytes -> wc, bytes as int (single byte < 0x100, pair = lead << 8 | trail), ascending.""" out = {} for b, wc in enumerate(single): if wc != NONE: out[b] = wc for lead in sorted(dbcs): for trail in sorted(dbcs[lead]): out[lead << 8 | trail] = dbcs[lead][trail] return out def default_inverse(decoded): """The encoder the runtime builds: the first byte sequence (ascending) that decodes to wc.""" inverse = {} for mb in sorted(decoded): inverse.setdefault(decoded[mb], mb) return inverse def encode_overrides(decoded, wctable): """WCTABLE entries the default inverse gets wrong: 0 round-trip, 1 best fit, 2 not encodable.""" inverse = default_inverse(decoded) out = [] for wc in sorted(set(inverse) | set(wctable)): want = wctable.get(wc) if want is None: out.append((wc, 0, 2)) elif decoded.get(want) != wc: out.append((wc, want, 1)) elif inverse.get(wc) != want: out.append((wc, want, 0)) return out def rows(values, per_line=16): return "\n".join("\t" + ", ".join(values[i:i + per_line]) + "," for i in range(0, len(values), per_line)) def emit(cp, single, leads, dbcs, wctable, mb_default): decoded = decode_map(single, dbcs) overrides = encode_overrides(decoded, wctable) # Sanity: the rebuilt encoder must reproduce WCTABLE exactly. inverse = default_inverse(decoded) for wc, mb, kind in overrides: if kind == 2: inverse.pop(wc, None) else: inverse[wc] = mb if inverse != wctable: raise SystemExit(f"cp{cp}: encoder rebuild does not match WCTABLE") name = f"kCp{cp}" text = [f"// Code page {cp}: {len(decoded)} byte sequences, {len(wctable)} WCTABLE entries."] text.append(f"const uint16_t {name}Single[256] = {{\n{rows([f'0x{v:04X}' for v in single])}\n}};") lead_ranges = "nullptr" dbcs_rows = "nullptr" dbcs_data = "nullptr" if leads: lead_ranges = f"{name}Leads" text.append(f"const uint8_t {name}Leads[] = {{" + ", ".join(f"0x{lo:02X}, 0x{hi:02X}" for lo, hi in leads) + "};") row_defs = [] data = [] for lead in sorted(dbcs): trails = dbcs[lead] lo, hi = min(trails), max(trails) row_defs.append(f"\t{{0x{lead:02X}, 0x{lo:02X}, 0x{hi:02X}, {len(data)}}},") data.extend(trails.get(t, NONE) for t in range(lo, hi + 1)) dbcs_rows = f"{name}Rows" dbcs_data = f"{name}Dbcs" text.append(f"const DbcsRow {name}Rows[] = {{\n" + "\n".join(row_defs) + "\n};") text.append(f"const uint16_t {name}Dbcs[] = {{\n{rows([f'0x{v:04X}' for v in data])}\n}};") text.append(f"const EncodeOverride {name}Overrides[] = {{\n" + rows([f"{{0x{wc:04X}, 0x{mb:04X}, {kind}}}" for wc, mb, kind in overrides], 6) + "\n};" if overrides else f"const EncodeOverride {name}Overrides[] = {{{{0, 0, 2}}}};") count = f"sizeof({name}Overrides) / sizeof({name}Overrides[0])" if overrides else "0" entry = (f"\t{{{cp}, 0x{mb_default:04X}, {name}Single, {lead_ranges}, " f"{len(leads)}, {dbcs_rows}, {len(dbcs) if leads else 0}, {dbcs_data}, " f"{name}Overrides, {count}}},") return "\n".join(text) + "\n", entry, len(overrides) def main(): parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) parser.add_argument("--cache", type=Path, default=REPO / "build/bestfit") parser.add_argument("--output", type=Path, default=REPO / "extension/src/codepage/codepage_tables.inc") args = parser.parse_args() body = [] entries = [] for cp in sorted(SHA256): parsed = parse(cp, fetch(cp, args.cache)) text, entry, n_over = emit(cp, *parsed) body.append(text) entries.append(entry) print(f"cp{cp}: {n_over} encode overrides", file=sys.stderr) header = ("// Generated by script/gen_codepage_tables.py from Microsoft's WindowsBestFit tables\n" "// (unicode.org MAPPINGS/VENDORS/MICSFT/WindowsBestFit). Do not edit.\n" "// 0xFFFF marks a byte / pair with no mapping. Included by codepage.cpp only.\n") table = "const CodePageTable kTables[] = {\n" + "\n".join(entries) + "\n};\n" args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(header + "\n" + "\n".join(body) + "\n" + table, encoding="ascii") print(f"wrote {args.output}", file=sys.stderr) if __name__ == "__main__": main()