codepage: replace iconv with bundled WindowsBestFit tables

mt_codepage implements code pages 874/932/936/949/950/1250-1258 from
Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py,
SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/
IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build
no longer depends on iconv. port.codepage checks every table against
its source file.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-28 18:45:49 +09:00
co-authored by Claude Opus 5.5
parent 07c101fdc8
commit 0b701d88c3
12 changed files with 7068 additions and 191 deletions
+227
View File
@@ -0,0 +1,227 @@
#!/usr/bin/env python3
"""Generate extension/src/codepage/codepage_tables.inc from Microsoft's WindowsBestFit tables.
The port emulates MultiByteToWideChar / WideCharToMultiByte off Windows. Instead of the host iconv
(Android's bionic has no GBK, and iconv tables differ from Windows in the corners), the conversions
use Microsoft's own code page definitions, published by unicode.org:
https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit<cp>.txt
Each file has an MBTABLE (single bytes), DBCSRANGE/DBCSTABLE (lead byte + trail byte) and a WCTABLE
(Unicode -> bytes, including the many-to-one "best fit" entries WideCharToMultiByte uses without
WC_NO_BEST_FIT_CHARS). The generated data keeps the byte -> Unicode side and only the WCTABLE entries
that differ from its first-match inverse, so the runtime can rebuild the encoder.
Usage: python3 script/gen_codepage_tables.py [--cache DIR] [--output FILE]
Downloads missing files into the cache (build/bestfit by default) and checks their SHA-256.
"""
import argparse
import hashlib
import re
import sys
import urllib.request
from pathlib import Path
REPO = Path(__file__).resolve().parent.parent
URL = "https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit{}.txt"
# Every code page 40250 EterLib/Util.cpp names (1252 included; CP_ACP maps to it).
SHA256 = {
874: "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd",
932: "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331",
936: "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae",
949: "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a",
950: "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a",
1250: "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b",
1251: "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7",
1252: "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9",
1253: "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e",
1254: "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a",
1255: "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a",
1256: "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007",
1257: "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366",
1258: "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1",
}
NONE = 0xFFFF
PAIR = re.compile(rb"^\s*0x([0-9a-fA-F]+)\s+0x([0-9a-fA-F]+)")
def fetch(cp: int, cache: Path) -> bytes:
path = cache / f"bestfit{cp}.txt"
if not path.is_file():
cache.mkdir(parents=True, exist_ok=True)
with urllib.request.urlopen(URL.format(cp)) as response:
path.write_bytes(response.read())
data = path.read_bytes()
digest = hashlib.sha256(data).hexdigest()
if digest != SHA256[cp]:
raise SystemExit(f"{path}: SHA-256 {digest} != {SHA256[cp]}")
return data
def parse(cp: int, data: bytes):
single = [NONE] * 256
leads = [] # [(lo, hi)]
dbcs = {} # lead -> {trail: wc}
wctable = {} # wc -> mb
mb_default = None
section = None
lead = None
for raw in data.splitlines():
line = raw.split(b";", 1)[0].strip()
if not line:
continue
head = line.split()[0]
if head == b"CODEPAGE":
if int(line.split()[1]) != cp:
raise SystemExit(f"bestfit{cp}.txt declares {line!r}")
continue
if head == b"CPINFO":
fields = line.split()
if int(fields[2], 16) != 0x3F:
raise SystemExit(f"cp{cp}: byte default char {fields[2]!r} is not '?'")
mb_default = int(fields[3], 16)
continue
if head in (b"MBTABLE", b"WCTABLE", b"DBCSRANGE"):
section = head
continue
if head == b"DBCSTABLE":
section = head
match = re.search(rb"LeadByte\s*=\s*0x([0-9a-fA-F]+)", raw)
lead = int(match.group(1), 16)
dbcs.setdefault(lead, {})
continue
if head == b"ENDCODEPAGE":
break
match = PAIR.match(line)
if not match:
raise SystemExit(f"cp{cp}: unexpected line {raw!r}")
a, b = int(match.group(1), 16), int(match.group(2), 16)
# cp932 lists its second lead range (0xe0-0xfc) after the first range's DBCSTABLEs.
if b"Lead Byte Range" in raw and section in (b"DBCSRANGE", b"DBCSTABLE"):
leads.append((a, b))
continue
if section == b"MBTABLE":
single[a] = b
elif section == b"DBCSRANGE":
leads.append((a, b))
elif section == b"DBCSTABLE":
dbcs[lead][a] = b
elif section == b"WCTABLE":
wctable[a] = b
if mb_default is None:
raise SystemExit(f"cp{cp}: no CPINFO")
for lead in dbcs:
if not any(lo <= lead <= hi for lo, hi in leads):
raise SystemExit(f"cp{cp}: DBCSTABLE 0x{lead:02x} outside the lead ranges {leads}")
return single, leads, dbcs, wctable, mb_default
def decode_map(single, dbcs):
"""bytes -> wc, bytes as int (single byte < 0x100, pair = lead << 8 | trail), ascending."""
out = {}
for b, wc in enumerate(single):
if wc != NONE:
out[b] = wc
for lead in sorted(dbcs):
for trail in sorted(dbcs[lead]):
out[lead << 8 | trail] = dbcs[lead][trail]
return out
def default_inverse(decoded):
"""The encoder the runtime builds: the first byte sequence (ascending) that decodes to wc."""
inverse = {}
for mb in sorted(decoded):
inverse.setdefault(decoded[mb], mb)
return inverse
def encode_overrides(decoded, wctable):
"""WCTABLE entries the default inverse gets wrong: 0 round-trip, 1 best fit, 2 not encodable."""
inverse = default_inverse(decoded)
out = []
for wc in sorted(set(inverse) | set(wctable)):
want = wctable.get(wc)
if want is None:
out.append((wc, 0, 2))
elif decoded.get(want) != wc:
out.append((wc, want, 1))
elif inverse.get(wc) != want:
out.append((wc, want, 0))
return out
def rows(values, per_line=16):
return "\n".join("\t" + ", ".join(values[i:i + per_line]) + "," for i in range(0, len(values), per_line))
def emit(cp, single, leads, dbcs, wctable, mb_default):
decoded = decode_map(single, dbcs)
overrides = encode_overrides(decoded, wctable)
# Sanity: the rebuilt encoder must reproduce WCTABLE exactly.
inverse = default_inverse(decoded)
for wc, mb, kind in overrides:
if kind == 2:
inverse.pop(wc, None)
else:
inverse[wc] = mb
if inverse != wctable:
raise SystemExit(f"cp{cp}: encoder rebuild does not match WCTABLE")
name = f"kCp{cp}"
text = [f"// Code page {cp}: {len(decoded)} byte sequences, {len(wctable)} WCTABLE entries."]
text.append(f"const uint16_t {name}Single[256] = {{\n{rows([f'0x{v:04X}' for v in single])}\n}};")
lead_ranges = "nullptr"
dbcs_rows = "nullptr"
dbcs_data = "nullptr"
if leads:
lead_ranges = f"{name}Leads"
text.append(f"const uint8_t {name}Leads[] = {{" + ", ".join(f"0x{lo:02X}, 0x{hi:02X}" for lo, hi in leads) + "};")
row_defs = []
data = []
for lead in sorted(dbcs):
trails = dbcs[lead]
lo, hi = min(trails), max(trails)
row_defs.append(f"\t{{0x{lead:02X}, 0x{lo:02X}, 0x{hi:02X}, {len(data)}}},")
data.extend(trails.get(t, NONE) for t in range(lo, hi + 1))
dbcs_rows = f"{name}Rows"
dbcs_data = f"{name}Dbcs"
text.append(f"const DbcsRow {name}Rows[] = {{\n" + "\n".join(row_defs) + "\n};")
text.append(f"const uint16_t {name}Dbcs[] = {{\n{rows([f'0x{v:04X}' for v in data])}\n}};")
text.append(f"const EncodeOverride {name}Overrides[] = {{\n"
+ rows([f"{{0x{wc:04X}, 0x{mb:04X}, {kind}}}" for wc, mb, kind in overrides], 6)
+ "\n};" if overrides else f"const EncodeOverride {name}Overrides[] = {{{{0, 0, 2}}}};")
count = f"sizeof({name}Overrides) / sizeof({name}Overrides[0])" if overrides else "0"
entry = (f"\t{{{cp}, 0x{mb_default:04X}, {name}Single, {lead_ranges}, "
f"{len(leads)}, {dbcs_rows}, {len(dbcs) if leads else 0}, {dbcs_data}, "
f"{name}Overrides, {count}}},")
return "\n".join(text) + "\n", entry, len(overrides)
def main():
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--cache", type=Path, default=REPO / "build/bestfit")
parser.add_argument("--output", type=Path, default=REPO / "extension/src/codepage/codepage_tables.inc")
args = parser.parse_args()
body = []
entries = []
for cp in sorted(SHA256):
parsed = parse(cp, fetch(cp, args.cache))
text, entry, n_over = emit(cp, *parsed)
body.append(text)
entries.append(entry)
print(f"cp{cp}: {n_over} encode overrides", file=sys.stderr)
header = ("// Generated by script/gen_codepage_tables.py from Microsoft's WindowsBestFit tables\n"
"// (unicode.org MAPPINGS/VENDORS/MICSFT/WindowsBestFit). Do not edit.\n"
"// 0xFFFF marks a byte / pair with no mapping. Included by codepage.cpp only.\n")
table = "const CodePageTable kTables[] = {\n" + "\n".join(entries) + "\n};\n"
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(header + "\n" + "\n".join(body) + "\n" + table, encoding="ascii")
print(f"wrote {args.output}", file=sys.stderr)
if __name__ == "__main__":
main()