mt_codepage implements code pages 874/932/936/949/950/1250-1258 from Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py, SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/ IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build no longer depends on iconv. port.codepage checks every table against its source file. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
228 lines
9.6 KiB
Python
228 lines
9.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Generate extension/src/codepage/codepage_tables.inc from Microsoft's WindowsBestFit tables.
|
|
|
|
The port emulates MultiByteToWideChar / WideCharToMultiByte off Windows. Instead of the host iconv
|
|
(Android's bionic has no GBK, and iconv tables differ from Windows in the corners), the conversions
|
|
use Microsoft's own code page definitions, published by unicode.org:
|
|
|
|
https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit<cp>.txt
|
|
|
|
Each file has an MBTABLE (single bytes), DBCSRANGE/DBCSTABLE (lead byte + trail byte) and a WCTABLE
|
|
(Unicode -> bytes, including the many-to-one "best fit" entries WideCharToMultiByte uses without
|
|
WC_NO_BEST_FIT_CHARS). The generated data keeps the byte -> Unicode side and only the WCTABLE entries
|
|
that differ from its first-match inverse, so the runtime can rebuild the encoder.
|
|
|
|
Usage: python3 script/gen_codepage_tables.py [--cache DIR] [--output FILE]
|
|
Downloads missing files into the cache (build/bestfit by default) and checks their SHA-256.
|
|
"""
|
|
import argparse
|
|
import hashlib
|
|
import re
|
|
import sys
|
|
import urllib.request
|
|
from pathlib import Path
|
|
|
|
REPO = Path(__file__).resolve().parent.parent
|
|
URL = "https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit{}.txt"
|
|
|
|
# Every code page 40250 EterLib/Util.cpp names (1252 included; CP_ACP maps to it).
|
|
SHA256 = {
|
|
874: "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd",
|
|
932: "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331",
|
|
936: "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae",
|
|
949: "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a",
|
|
950: "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a",
|
|
1250: "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b",
|
|
1251: "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7",
|
|
1252: "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9",
|
|
1253: "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e",
|
|
1254: "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a",
|
|
1255: "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a",
|
|
1256: "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007",
|
|
1257: "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366",
|
|
1258: "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1",
|
|
}
|
|
|
|
NONE = 0xFFFF
|
|
PAIR = re.compile(rb"^\s*0x([0-9a-fA-F]+)\s+0x([0-9a-fA-F]+)")
|
|
|
|
|
|
def fetch(cp: int, cache: Path) -> bytes:
|
|
path = cache / f"bestfit{cp}.txt"
|
|
if not path.is_file():
|
|
cache.mkdir(parents=True, exist_ok=True)
|
|
with urllib.request.urlopen(URL.format(cp)) as response:
|
|
path.write_bytes(response.read())
|
|
data = path.read_bytes()
|
|
digest = hashlib.sha256(data).hexdigest()
|
|
if digest != SHA256[cp]:
|
|
raise SystemExit(f"{path}: SHA-256 {digest} != {SHA256[cp]}")
|
|
return data
|
|
|
|
|
|
def parse(cp: int, data: bytes):
|
|
single = [NONE] * 256
|
|
leads = [] # [(lo, hi)]
|
|
dbcs = {} # lead -> {trail: wc}
|
|
wctable = {} # wc -> mb
|
|
mb_default = None
|
|
section = None
|
|
lead = None
|
|
for raw in data.splitlines():
|
|
line = raw.split(b";", 1)[0].strip()
|
|
if not line:
|
|
continue
|
|
head = line.split()[0]
|
|
if head == b"CODEPAGE":
|
|
if int(line.split()[1]) != cp:
|
|
raise SystemExit(f"bestfit{cp}.txt declares {line!r}")
|
|
continue
|
|
if head == b"CPINFO":
|
|
fields = line.split()
|
|
if int(fields[2], 16) != 0x3F:
|
|
raise SystemExit(f"cp{cp}: byte default char {fields[2]!r} is not '?'")
|
|
mb_default = int(fields[3], 16)
|
|
continue
|
|
if head in (b"MBTABLE", b"WCTABLE", b"DBCSRANGE"):
|
|
section = head
|
|
continue
|
|
if head == b"DBCSTABLE":
|
|
section = head
|
|
match = re.search(rb"LeadByte\s*=\s*0x([0-9a-fA-F]+)", raw)
|
|
lead = int(match.group(1), 16)
|
|
dbcs.setdefault(lead, {})
|
|
continue
|
|
if head == b"ENDCODEPAGE":
|
|
break
|
|
match = PAIR.match(line)
|
|
if not match:
|
|
raise SystemExit(f"cp{cp}: unexpected line {raw!r}")
|
|
a, b = int(match.group(1), 16), int(match.group(2), 16)
|
|
# cp932 lists its second lead range (0xe0-0xfc) after the first range's DBCSTABLEs.
|
|
if b"Lead Byte Range" in raw and section in (b"DBCSRANGE", b"DBCSTABLE"):
|
|
leads.append((a, b))
|
|
continue
|
|
if section == b"MBTABLE":
|
|
single[a] = b
|
|
elif section == b"DBCSRANGE":
|
|
leads.append((a, b))
|
|
elif section == b"DBCSTABLE":
|
|
dbcs[lead][a] = b
|
|
elif section == b"WCTABLE":
|
|
wctable[a] = b
|
|
if mb_default is None:
|
|
raise SystemExit(f"cp{cp}: no CPINFO")
|
|
for lead in dbcs:
|
|
if not any(lo <= lead <= hi for lo, hi in leads):
|
|
raise SystemExit(f"cp{cp}: DBCSTABLE 0x{lead:02x} outside the lead ranges {leads}")
|
|
return single, leads, dbcs, wctable, mb_default
|
|
|
|
|
|
def decode_map(single, dbcs):
|
|
"""bytes -> wc, bytes as int (single byte < 0x100, pair = lead << 8 | trail), ascending."""
|
|
out = {}
|
|
for b, wc in enumerate(single):
|
|
if wc != NONE:
|
|
out[b] = wc
|
|
for lead in sorted(dbcs):
|
|
for trail in sorted(dbcs[lead]):
|
|
out[lead << 8 | trail] = dbcs[lead][trail]
|
|
return out
|
|
|
|
|
|
def default_inverse(decoded):
|
|
"""The encoder the runtime builds: the first byte sequence (ascending) that decodes to wc."""
|
|
inverse = {}
|
|
for mb in sorted(decoded):
|
|
inverse.setdefault(decoded[mb], mb)
|
|
return inverse
|
|
|
|
|
|
def encode_overrides(decoded, wctable):
|
|
"""WCTABLE entries the default inverse gets wrong: 0 round-trip, 1 best fit, 2 not encodable."""
|
|
inverse = default_inverse(decoded)
|
|
out = []
|
|
for wc in sorted(set(inverse) | set(wctable)):
|
|
want = wctable.get(wc)
|
|
if want is None:
|
|
out.append((wc, 0, 2))
|
|
elif decoded.get(want) != wc:
|
|
out.append((wc, want, 1))
|
|
elif inverse.get(wc) != want:
|
|
out.append((wc, want, 0))
|
|
return out
|
|
|
|
|
|
def rows(values, per_line=16):
|
|
return "\n".join("\t" + ", ".join(values[i:i + per_line]) + "," for i in range(0, len(values), per_line))
|
|
|
|
|
|
def emit(cp, single, leads, dbcs, wctable, mb_default):
|
|
decoded = decode_map(single, dbcs)
|
|
overrides = encode_overrides(decoded, wctable)
|
|
# Sanity: the rebuilt encoder must reproduce WCTABLE exactly.
|
|
inverse = default_inverse(decoded)
|
|
for wc, mb, kind in overrides:
|
|
if kind == 2:
|
|
inverse.pop(wc, None)
|
|
else:
|
|
inverse[wc] = mb
|
|
if inverse != wctable:
|
|
raise SystemExit(f"cp{cp}: encoder rebuild does not match WCTABLE")
|
|
|
|
name = f"kCp{cp}"
|
|
text = [f"// Code page {cp}: {len(decoded)} byte sequences, {len(wctable)} WCTABLE entries."]
|
|
text.append(f"const uint16_t {name}Single[256] = {{\n{rows([f'0x{v:04X}' for v in single])}\n}};")
|
|
lead_ranges = "nullptr"
|
|
dbcs_rows = "nullptr"
|
|
dbcs_data = "nullptr"
|
|
if leads:
|
|
lead_ranges = f"{name}Leads"
|
|
text.append(f"const uint8_t {name}Leads[] = {{" + ", ".join(f"0x{lo:02X}, 0x{hi:02X}" for lo, hi in leads) + "};")
|
|
row_defs = []
|
|
data = []
|
|
for lead in sorted(dbcs):
|
|
trails = dbcs[lead]
|
|
lo, hi = min(trails), max(trails)
|
|
row_defs.append(f"\t{{0x{lead:02X}, 0x{lo:02X}, 0x{hi:02X}, {len(data)}}},")
|
|
data.extend(trails.get(t, NONE) for t in range(lo, hi + 1))
|
|
dbcs_rows = f"{name}Rows"
|
|
dbcs_data = f"{name}Dbcs"
|
|
text.append(f"const DbcsRow {name}Rows[] = {{\n" + "\n".join(row_defs) + "\n};")
|
|
text.append(f"const uint16_t {name}Dbcs[] = {{\n{rows([f'0x{v:04X}' for v in data])}\n}};")
|
|
text.append(f"const EncodeOverride {name}Overrides[] = {{\n"
|
|
+ rows([f"{{0x{wc:04X}, 0x{mb:04X}, {kind}}}" for wc, mb, kind in overrides], 6)
|
|
+ "\n};" if overrides else f"const EncodeOverride {name}Overrides[] = {{{{0, 0, 2}}}};")
|
|
count = f"sizeof({name}Overrides) / sizeof({name}Overrides[0])" if overrides else "0"
|
|
entry = (f"\t{{{cp}, 0x{mb_default:04X}, {name}Single, {lead_ranges}, "
|
|
f"{len(leads)}, {dbcs_rows}, {len(dbcs) if leads else 0}, {dbcs_data}, "
|
|
f"{name}Overrides, {count}}},")
|
|
return "\n".join(text) + "\n", entry, len(overrides)
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
|
parser.add_argument("--cache", type=Path, default=REPO / "build/bestfit")
|
|
parser.add_argument("--output", type=Path, default=REPO / "extension/src/codepage/codepage_tables.inc")
|
|
args = parser.parse_args()
|
|
|
|
body = []
|
|
entries = []
|
|
for cp in sorted(SHA256):
|
|
parsed = parse(cp, fetch(cp, args.cache))
|
|
text, entry, n_over = emit(cp, *parsed)
|
|
body.append(text)
|
|
entries.append(entry)
|
|
print(f"cp{cp}: {n_over} encode overrides", file=sys.stderr)
|
|
header = ("// Generated by script/gen_codepage_tables.py from Microsoft's WindowsBestFit tables\n"
|
|
"// (unicode.org MAPPINGS/VENDORS/MICSFT/WindowsBestFit). Do not edit.\n"
|
|
"// 0xFFFF marks a byte / pair with no mapping. Included by codepage.cpp only.\n")
|
|
table = "const CodePageTable kTables[] = {\n" + "\n".join(entries) + "\n};\n"
|
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
|
args.output.write_text(header + "\n" + "\n".join(body) + "\n" + table, encoding="ascii")
|
|
print(f"wrote {args.output}", file=sys.stderr)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|