codepage: replace iconv with bundled WindowsBestFit tables
mt_codepage implements code pages 874/932/936/949/950/1250-1258 from Microsoft's bestfit<cp>.txt sources (script/gen_codepage_tables.py, SHA-256 pinned). Win32Crt's MultiByteToWideChar/WideCharToMultiByte/ IsDBCSLeadByteEx and net/text_codec use it, so the Android API 24 build no longer depends on iconv. port.codepage checks every table against its source file. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
07c101fdc8
commit
0b701d88c3
@@ -0,0 +1,227 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate extension/src/codepage/codepage_tables.inc from Microsoft's WindowsBestFit tables.
|
||||
|
||||
The port emulates MultiByteToWideChar / WideCharToMultiByte off Windows. Instead of the host iconv
|
||||
(Android's bionic has no GBK, and iconv tables differ from Windows in the corners), the conversions
|
||||
use Microsoft's own code page definitions, published by unicode.org:
|
||||
|
||||
https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit<cp>.txt
|
||||
|
||||
Each file has an MBTABLE (single bytes), DBCSRANGE/DBCSTABLE (lead byte + trail byte) and a WCTABLE
|
||||
(Unicode -> bytes, including the many-to-one "best fit" entries WideCharToMultiByte uses without
|
||||
WC_NO_BEST_FIT_CHARS). The generated data keeps the byte -> Unicode side and only the WCTABLE entries
|
||||
that differ from its first-match inverse, so the runtime can rebuild the encoder.
|
||||
|
||||
Usage: python3 script/gen_codepage_tables.py [--cache DIR] [--output FILE]
|
||||
Downloads missing files into the cache (build/bestfit by default) and checks their SHA-256.
|
||||
"""
|
||||
import argparse
|
||||
import hashlib
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
URL = "https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/bestfit{}.txt"
|
||||
|
||||
# Every code page 40250 EterLib/Util.cpp names (1252 included; CP_ACP maps to it).
|
||||
SHA256 = {
|
||||
874: "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd",
|
||||
932: "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331",
|
||||
936: "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae",
|
||||
949: "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a",
|
||||
950: "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a",
|
||||
1250: "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b",
|
||||
1251: "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7",
|
||||
1252: "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9",
|
||||
1253: "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e",
|
||||
1254: "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a",
|
||||
1255: "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a",
|
||||
1256: "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007",
|
||||
1257: "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366",
|
||||
1258: "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1",
|
||||
}
|
||||
|
||||
NONE = 0xFFFF
|
||||
PAIR = re.compile(rb"^\s*0x([0-9a-fA-F]+)\s+0x([0-9a-fA-F]+)")
|
||||
|
||||
|
||||
def fetch(cp: int, cache: Path) -> bytes:
|
||||
path = cache / f"bestfit{cp}.txt"
|
||||
if not path.is_file():
|
||||
cache.mkdir(parents=True, exist_ok=True)
|
||||
with urllib.request.urlopen(URL.format(cp)) as response:
|
||||
path.write_bytes(response.read())
|
||||
data = path.read_bytes()
|
||||
digest = hashlib.sha256(data).hexdigest()
|
||||
if digest != SHA256[cp]:
|
||||
raise SystemExit(f"{path}: SHA-256 {digest} != {SHA256[cp]}")
|
||||
return data
|
||||
|
||||
|
||||
def parse(cp: int, data: bytes):
|
||||
single = [NONE] * 256
|
||||
leads = [] # [(lo, hi)]
|
||||
dbcs = {} # lead -> {trail: wc}
|
||||
wctable = {} # wc -> mb
|
||||
mb_default = None
|
||||
section = None
|
||||
lead = None
|
||||
for raw in data.splitlines():
|
||||
line = raw.split(b";", 1)[0].strip()
|
||||
if not line:
|
||||
continue
|
||||
head = line.split()[0]
|
||||
if head == b"CODEPAGE":
|
||||
if int(line.split()[1]) != cp:
|
||||
raise SystemExit(f"bestfit{cp}.txt declares {line!r}")
|
||||
continue
|
||||
if head == b"CPINFO":
|
||||
fields = line.split()
|
||||
if int(fields[2], 16) != 0x3F:
|
||||
raise SystemExit(f"cp{cp}: byte default char {fields[2]!r} is not '?'")
|
||||
mb_default = int(fields[3], 16)
|
||||
continue
|
||||
if head in (b"MBTABLE", b"WCTABLE", b"DBCSRANGE"):
|
||||
section = head
|
||||
continue
|
||||
if head == b"DBCSTABLE":
|
||||
section = head
|
||||
match = re.search(rb"LeadByte\s*=\s*0x([0-9a-fA-F]+)", raw)
|
||||
lead = int(match.group(1), 16)
|
||||
dbcs.setdefault(lead, {})
|
||||
continue
|
||||
if head == b"ENDCODEPAGE":
|
||||
break
|
||||
match = PAIR.match(line)
|
||||
if not match:
|
||||
raise SystemExit(f"cp{cp}: unexpected line {raw!r}")
|
||||
a, b = int(match.group(1), 16), int(match.group(2), 16)
|
||||
# cp932 lists its second lead range (0xe0-0xfc) after the first range's DBCSTABLEs.
|
||||
if b"Lead Byte Range" in raw and section in (b"DBCSRANGE", b"DBCSTABLE"):
|
||||
leads.append((a, b))
|
||||
continue
|
||||
if section == b"MBTABLE":
|
||||
single[a] = b
|
||||
elif section == b"DBCSRANGE":
|
||||
leads.append((a, b))
|
||||
elif section == b"DBCSTABLE":
|
||||
dbcs[lead][a] = b
|
||||
elif section == b"WCTABLE":
|
||||
wctable[a] = b
|
||||
if mb_default is None:
|
||||
raise SystemExit(f"cp{cp}: no CPINFO")
|
||||
for lead in dbcs:
|
||||
if not any(lo <= lead <= hi for lo, hi in leads):
|
||||
raise SystemExit(f"cp{cp}: DBCSTABLE 0x{lead:02x} outside the lead ranges {leads}")
|
||||
return single, leads, dbcs, wctable, mb_default
|
||||
|
||||
|
||||
def decode_map(single, dbcs):
|
||||
"""bytes -> wc, bytes as int (single byte < 0x100, pair = lead << 8 | trail), ascending."""
|
||||
out = {}
|
||||
for b, wc in enumerate(single):
|
||||
if wc != NONE:
|
||||
out[b] = wc
|
||||
for lead in sorted(dbcs):
|
||||
for trail in sorted(dbcs[lead]):
|
||||
out[lead << 8 | trail] = dbcs[lead][trail]
|
||||
return out
|
||||
|
||||
|
||||
def default_inverse(decoded):
|
||||
"""The encoder the runtime builds: the first byte sequence (ascending) that decodes to wc."""
|
||||
inverse = {}
|
||||
for mb in sorted(decoded):
|
||||
inverse.setdefault(decoded[mb], mb)
|
||||
return inverse
|
||||
|
||||
|
||||
def encode_overrides(decoded, wctable):
|
||||
"""WCTABLE entries the default inverse gets wrong: 0 round-trip, 1 best fit, 2 not encodable."""
|
||||
inverse = default_inverse(decoded)
|
||||
out = []
|
||||
for wc in sorted(set(inverse) | set(wctable)):
|
||||
want = wctable.get(wc)
|
||||
if want is None:
|
||||
out.append((wc, 0, 2))
|
||||
elif decoded.get(want) != wc:
|
||||
out.append((wc, want, 1))
|
||||
elif inverse.get(wc) != want:
|
||||
out.append((wc, want, 0))
|
||||
return out
|
||||
|
||||
|
||||
def rows(values, per_line=16):
|
||||
return "\n".join("\t" + ", ".join(values[i:i + per_line]) + "," for i in range(0, len(values), per_line))
|
||||
|
||||
|
||||
def emit(cp, single, leads, dbcs, wctable, mb_default):
|
||||
decoded = decode_map(single, dbcs)
|
||||
overrides = encode_overrides(decoded, wctable)
|
||||
# Sanity: the rebuilt encoder must reproduce WCTABLE exactly.
|
||||
inverse = default_inverse(decoded)
|
||||
for wc, mb, kind in overrides:
|
||||
if kind == 2:
|
||||
inverse.pop(wc, None)
|
||||
else:
|
||||
inverse[wc] = mb
|
||||
if inverse != wctable:
|
||||
raise SystemExit(f"cp{cp}: encoder rebuild does not match WCTABLE")
|
||||
|
||||
name = f"kCp{cp}"
|
||||
text = [f"// Code page {cp}: {len(decoded)} byte sequences, {len(wctable)} WCTABLE entries."]
|
||||
text.append(f"const uint16_t {name}Single[256] = {{\n{rows([f'0x{v:04X}' for v in single])}\n}};")
|
||||
lead_ranges = "nullptr"
|
||||
dbcs_rows = "nullptr"
|
||||
dbcs_data = "nullptr"
|
||||
if leads:
|
||||
lead_ranges = f"{name}Leads"
|
||||
text.append(f"const uint8_t {name}Leads[] = {{" + ", ".join(f"0x{lo:02X}, 0x{hi:02X}" for lo, hi in leads) + "};")
|
||||
row_defs = []
|
||||
data = []
|
||||
for lead in sorted(dbcs):
|
||||
trails = dbcs[lead]
|
||||
lo, hi = min(trails), max(trails)
|
||||
row_defs.append(f"\t{{0x{lead:02X}, 0x{lo:02X}, 0x{hi:02X}, {len(data)}}},")
|
||||
data.extend(trails.get(t, NONE) for t in range(lo, hi + 1))
|
||||
dbcs_rows = f"{name}Rows"
|
||||
dbcs_data = f"{name}Dbcs"
|
||||
text.append(f"const DbcsRow {name}Rows[] = {{\n" + "\n".join(row_defs) + "\n};")
|
||||
text.append(f"const uint16_t {name}Dbcs[] = {{\n{rows([f'0x{v:04X}' for v in data])}\n}};")
|
||||
text.append(f"const EncodeOverride {name}Overrides[] = {{\n"
|
||||
+ rows([f"{{0x{wc:04X}, 0x{mb:04X}, {kind}}}" for wc, mb, kind in overrides], 6)
|
||||
+ "\n};" if overrides else f"const EncodeOverride {name}Overrides[] = {{{{0, 0, 2}}}};")
|
||||
count = f"sizeof({name}Overrides) / sizeof({name}Overrides[0])" if overrides else "0"
|
||||
entry = (f"\t{{{cp}, 0x{mb_default:04X}, {name}Single, {lead_ranges}, "
|
||||
f"{len(leads)}, {dbcs_rows}, {len(dbcs) if leads else 0}, {dbcs_data}, "
|
||||
f"{name}Overrides, {count}}},")
|
||||
return "\n".join(text) + "\n", entry, len(overrides)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
parser.add_argument("--cache", type=Path, default=REPO / "build/bestfit")
|
||||
parser.add_argument("--output", type=Path, default=REPO / "extension/src/codepage/codepage_tables.inc")
|
||||
args = parser.parse_args()
|
||||
|
||||
body = []
|
||||
entries = []
|
||||
for cp in sorted(SHA256):
|
||||
parsed = parse(cp, fetch(cp, args.cache))
|
||||
text, entry, n_over = emit(cp, *parsed)
|
||||
body.append(text)
|
||||
entries.append(entry)
|
||||
print(f"cp{cp}: {n_over} encode overrides", file=sys.stderr)
|
||||
header = ("// Generated by script/gen_codepage_tables.py from Microsoft's WindowsBestFit tables\n"
|
||||
"// (unicode.org MAPPINGS/VENDORS/MICSFT/WindowsBestFit). Do not edit.\n"
|
||||
"// 0xFFFF marks a byte / pair with no mapping. Included by codepage.cpp only.\n")
|
||||
table = "const CodePageTable kTables[] = {\n" + "\n".join(entries) + "\n};\n"
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(header + "\n" + "\n".join(body) + "\n" + table, encoding="ascii")
|
||||
print(f"wrote {args.output}", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user