#!/usr/bin/env python3 """Build a CP936 Chinese locale on top of the 40250 English client resources. The Longju locale is an older client version. Only keyed UI text with compatible format arguments is overlaid; 40250 proto, scripts, and UI layouts stay intact. The files are staged in build/zh-locale/locale-src/zh and packed like the original locale packs (eterpack_pack: COMPRESS entries, encrypted index) into locale_zh.eix/.epk. The output Client has a pack/ of its own: links to the original packs, the zh pack and an Index registering locale/zh/ -> locale_zh. No loose locale/zh is left next to it, since the client searches loose files first. """ import argparse import json import re import shutil import subprocess from pathlib import Path REPO = Path(__file__).resolve().parents[1] DEFAULT_EN = REPO.parent / "40250/Server Client TMP4/Client/Eternexus/locale_en/locale/en" DEFAULT_ZH = REPO.parent / "server_extracted/client_locale/longju/locale/taiwan" DEFAULT_OUTPUT = REPO / "build/zh-locale/Client" BUILD_DIRS = ("build/release", "build/native") LOCALE_CFG = "10002 936 zh\n" DEFAULT_NAMES = REPO / "data/zh_40250" FORMAT = re.compile(r"%(?:\([^)]*\))?[-+#0 ]*\d*(?:\.\d+)?[diouxXeEfFgGcrs%]") KEYED_FILES = ("locale_game.txt", "locale_interface.txt") # Image folders whose labels are drawn text (the character window's tabs and stat labels). Each is taken # from Longju as a whole: its .sub files carry their own coordinates into the folder's atlas. IMAGE_DIRS = ("ui/windows",) # skilldesc.txt columns taken from Longju: names, description, conditions (2-8), affect texts (17/20/23). SKILL_COLUMNS = (2, 3, 4, 5, 6, 7, 8, 17, 20, 23) # vnum-keyed 40250 files whose rows an overlay (_zh_overlay.txt) replaces whole: items the server # gives a new use (tools/server/hunt-set) get a new icon and description. ROW_FILES = ("item_list.txt", "itemdesc.txt") # itemdesc.txt columns (0-based: description, summary) translated by their English text (itemdesc_translate_zh.tsv). ITEMDESC_COLUMNS = (2, 3) # The zh client keeps 40250's Western tooltip path (uitooltip.SplitDescription): lines break at spaces, at most # 35 bytes, and a token "|x" starts a new line. Chinese is wrapped here to the Eastern tooltip width instead # (DESC_DEFAULT_MAX_COLS = 26 bytes, a CJK character is 2) and each line after the first starts with "|". DESC_COLUMNS = 26 BREAK_AFTER = ",。、;:!?)”" def keyed_lines(path: Path, encoding: str): lines = path.read_bytes().decode(encoding).splitlines() values = {} for line in lines: parts = line.split("\t") if len(parts) >= 2 and parts[0]: values[parts[0]] = parts[1] return lines, values def name_rows(path: Path): rows = {} if path.is_file(): for line in path.read_text(encoding="utf-8").splitlines(): parts = line.split("\t", 1) if len(parts) == 2 and parts[0].isdigit(): rows[parts[0]] = parts[1] return rows def convert_names(source: Path, destination: Path, overlay: Path = None): """The server name snapshot, with the reviewed translations of its English names laid over it.""" names = name_rows(source) overlaid = {vnum: name for vnum, name in name_rows(overlay).items() if vnum in names} if overlay else {} names.update(overlaid) converted = 0 unrepresentable = [] rows = [] for vnum, name in names.items(): try: rows.append(f"{vnum}\t{name}\n".encode("gbk")) converted += 1 except UnicodeEncodeError: unrepresentable.append(vnum) destination.write_bytes(b"".join(rows)) return {"converted": converted, "overlaid": len(overlaid), "unrepresentable_vnums": unrepresentable} def wrap_lines(text: str, columns: int): atoms = re.findall(r"[\x21-\x7e]+|\s+|.", text.strip()) lines, line, width, cut = [], [], 0, None for atom in atoms: size = len(atom.encode("gbk")) if line and width + size > columns and atom not in BREAK_AFTER: rest = [] if cut is not None and cut >= len(line) // 2: rest, line = line[cut:], line[:cut] lines.append("".join(line).strip()) while rest and rest[0].isspace(): rest.pop(0) line, cut = rest, None width = sum(len(a.encode("gbk")) for a in line) if atom.isspace() and not line: continue line.append(atom) width += size if atom in BREAK_AFTER: cut = len(line) if line: lines.append("".join(line).strip()) return lines def wrap_description(text: str) -> str: """Chinese text as SplitDescription lines of at most DESC_COLUMNS bytes. ASCII runs (numbers, words) are not split; a line ends after punctuation when one is in its second half, and punctuation never starts a line (it may hang one character past the width, which the 35-byte Western limit still keeps on the line). A last line of one or two characters is avoided by narrowing the lines when that needs no extra line.""" if all(ord(c) < 128 for c in text) or "|" in text: return text lines = wrap_lines(text, DESC_COLUMNS) for columns in range(DESC_COLUMNS - 2, DESC_COLUMNS - 9, -2): if len(lines) < 2 or len(lines[-1].encode("gbk")) > 6: break narrower = wrap_lines(text, columns) if len(narrower) == len(lines): lines = narrower return " |".join(lines) def translate_itemdesc(path: Path, table: Path): """itemdesc.txt with its description and summary columns replaced from the English→Chinese table (UTF-8, EnglishChinese) and wrapped; untranslated cells keep the English text.""" chinese = {} for line in table.read_text(encoding="utf-8").splitlines(): if line and not line.startswith("#"): english, text = line.split("\t") chinese[english] = wrap_description(text) rows, translated, english_rows = [], 0, [] for line in path.read_bytes().decode("cp1252").split("\n"): parts = line.split("\t") if parts[0].strip().isdigit(): left = False for column in ITEMDESC_COLUMNS: text = parts[column].strip() if column < len(parts) else "" if not text: continue if text in chinese: parts[column] = chinese[text] + ("\r" if parts[column].endswith("\r") else "") translated += 1 else: left = True if left: english_rows.append(parts[0].strip()) rows.append("\t".join(parts)) path.write_bytes("\n".join(rows).encode("gbk")) return {"translated_cells": translated, "english_rows": english_rows} def overlay_rows(path: Path, overlay: Path, wrap_columns=()): """Replace the rows of a vnum-keyed file by the overlay's (UTF-8, whole rows, no header); other rows keep their bytes (the English files are cp1252). wrap_columns (0-based) are wrapped like translated text.""" rows = {} for line in overlay.read_text(encoding="utf-8").splitlines(): if line and not line.startswith("#"): parts = line.split("\t") for column in wrap_columns: if column < len(parts): parts[column] = wrap_description(parts[column]) rows[parts[0]] = "\t".join(parts).encode("gbk") lines = path.read_bytes().split(b"\n") replaced = [] for index, line in enumerate(lines): key = line.split(b"\t", 1)[0].decode("ascii", "replace") if key in rows: lines[index] = rows[key] + (b"\r" if line.endswith(b"\r") else b"") replaced.append(key) missing = sorted(rows.keys() - set(replaced)) if missing: raise ValueError(f"{path.name}: overlay rows {missing} not in the file") path.write_bytes(b"\n".join(lines)) return {"replaced_rows": replaced} def merge_skilldesc(en: Path, zh: Path, destination: Path, overlay: Path = None): """40250 rows with Longju's text columns; an affect text keeps the English one unless its format arguments match, since the client passes it to _snprintf with the min/max values.""" chinese = {} for line in zh.read_bytes().decode("gb18030").splitlines(): parts = line.split("\t") if parts[0].isdigit(): chinese[parts[0]] = parts reviewed = {} if overlay and overlay.is_file(): for line in overlay.read_text(encoding="utf-8").splitlines()[1:]: vnum, column, text = line.split("\t") reviewed[(vnum, int(column))] = text rows, translated, incompatible = [], 0, [] for line in en.read_bytes().decode("cp1252").splitlines(): parts = line.split("\t") if parts[0].isdigit(): source = chinese.get(parts[0], []) for column in SKILL_COLUMNS: if column >= len(parts): continue text = reviewed.get((parts[0], column)) if text is None and column < len(source) and source[column]: text = source[column] if text is None: continue if FORMAT.findall(parts[column]) != FORMAT.findall(text): incompatible.append(f"{parts[0]}:{column}") continue parts[column] = text translated += 1 rows.append("\t".join(parts)) destination.write_bytes(("\n".join(rows) + "\n").encode("gbk")) return {"translated_cells": translated, "incompatible_format_cells": incompatible} def tool(name: str) -> Path: for build_dir in BUILD_DIRS: path = REPO / build_dir / "src/port" / name if path.is_file(): return path return REPO / BUILD_DIRS[0] / "src/port" / name def register(index: bytes) -> bytes: """pack/Index with locale/zh/ -> locale_zh added before the locale/en/ pair (CRLF, as the original).""" lines = index.split(b"\r\n") if b"locale/zh/" in lines: return index at = lines.index(b"locale/en/") if at % 2 != 1: raise ValueError("pack/Index: locale/en/ is not a folder line") lines[at:at] = [b"locale/zh/", b"locale_zh"] return b"\r\n".join(lines) def make_pack_dir(client_base: Path, pack_dir: Path, locale_dir: Path, pack_tool: Path): source = client_base / "pack" if pack_dir.is_symlink(): pack_dir.unlink() # earlier builds linked the whole directory pack_dir.mkdir(parents=True, exist_ok=True) for entry in source.iterdir(): if entry.name == "Index" or entry.name.lower().startswith("locale_zh."): continue target = pack_dir / entry.name if target.is_symlink() or target.exists(): target.unlink() target.symlink_to(entry.resolve()) (pack_dir / "Index").write_bytes(register((source / "Index").read_bytes())) result = subprocess.run([str(pack_tool), str(pack_dir / "locale_zh"), str(locale_dir), "locale/zh/"], check=True, capture_output=True, text=True) return result.stdout.strip() def build(en: Path, zh: Path, output: Path, client_base: Path, proto_tool: Path = None, names_dir: Path = None, pack_tool: Path = None): """names_dir holds the server name snapshot and the reviewed *_zh_overlay.txt translations.""" if not (en / "locale_game.txt").is_file() or not (zh / "locale_game.txt").is_file(): raise ValueError("English 40250 and Chinese Longju locale directories are required") if pack_tool is None or not pack_tool.is_file(): raise ValueError("build eterpack_pack (cmake --build --target eterpack_pack)") locale_dir = output.parent / "locale-src/zh" if locale_dir.exists(): shutil.rmtree(locale_dir) shutil.copytree(en, locale_dir) loose = output / "locale/zh" if loose.exists(): shutil.rmtree(loose) # earlier builds ran the locale from loose files if not any((output / "locale").iterdir()): (output / "locale").rmdir() report = {"english_source": str(en), "chinese_source": str(zh), "files": {}} for filename in KEYED_FILES: english_lines, english = keyed_lines(en / filename, "cp1252") _, chinese = keyed_lines(zh / filename, "gbk") # Reviewed translations of the keys Longju lacks or formats differently; they win over Longju. overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None if overlay and overlay.is_file(): _, reviewed = keyed_lines(overlay, "utf-8") chinese.update(reviewed) translated = [] incompatible = [] merged = [] for line in english_lines: parts = line.split("\t") if len(parts) >= 2 and parts[0] in chinese: key = parts[0] candidate = chinese[key] if FORMAT.findall(parts[1]) == FORMAT.findall(candidate): parts[1] = candidate translated.append(key) else: incompatible.append(key) line = "\t".join(parts) merged.append(line) (locale_dir / filename).write_bytes(("\n".join(merged) + "\n").encode("gbk")) report["files"][filename] = { "reference_keys": len(english), "translated_keys": len(translated), "english_fallback_keys": sorted(english.keys() - set(translated)), "incompatible_format_keys": sorted(incompatible), "chinese_only_keys": sorted(chinese.keys() - english.keys()), } # These files are prose, not versioned game data. Keep the 40250 UI layouts # and prototype binaries copied above; only replace their explanatory text. for filename in ( "jobdesc_warrior.txt", "jobdesc_assassin.txt", "jobdesc_sura.txt", "jobdesc_shaman.txt", "empiredesc_a.txt", "empiredesc_b.txt", "empiredesc_c.txt", ): if (zh / filename).is_file(): shutil.copyfile(zh / filename, locale_dir / filename) if (zh / "skilldesc.txt").is_file(): report["files"]["skilldesc.txt"] = merge_skilldesc( en / "skilldesc.txt", zh / "skilldesc.txt", locale_dir / "skilldesc.txt", names_dir / "skilldesc_zh_overlay.txt" if names_dir else None) table = names_dir / "itemdesc_translate_zh.tsv" if names_dir else None if table and table.is_file(): report["files"]["itemdesc.txt translation"] = translate_itemdesc(locale_dir / "itemdesc.txt", table) for filename in ROW_FILES: overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None if overlay and overlay.is_file(): wrap = ITEMDESC_COLUMNS if filename == "itemdesc.txt" else () report["files"][filename] = overlay_rows(locale_dir / filename, overlay, wrap) for folder in IMAGE_DIRS: english_files = {path.name for path in (en / folder).iterdir()} chinese_files = {path.name for path in (zh / folder).iterdir()} if not english_files <= chinese_files: raise ValueError(f"{folder}: Longju lacks {sorted(english_files - chinese_files)}") for name in sorted(english_files): shutil.copyfile(zh / folder / name, locale_dir / folder / name) report["files"][folder] = {"replaced_images": len(english_files)} if proto_tool is not None: if not proto_tool.is_file() or names_dir is None: raise ValueError("build locale_proto_patch and provide the Chinese name tables") item_names = output.parent / "item_names_gbk.txt" mob_names = output.parent / "mob_names_gbk.txt" item_conversion = convert_names(names_dir / "item_names_zh.txt", item_names, names_dir / "item_names_zh_overlay.txt") mob_conversion = convert_names(names_dir / "mob_names_zh.txt", mob_names, names_dir / "mob_names_zh_overlay.txt") # item_limits_server.tsv: item limits changed on our server, so the tooltip matches what it enforces. limits = names_dir / "item_limits_server.tsv" result = subprocess.run( [str(proto_tool), str(en / "item_proto"), str(en / "mob_proto"), str(item_names), str(mob_names), str(locale_dir / "item_proto"), str(locale_dir / "mob_proto")] + ([str(limits)] if limits.is_file() else []), check=True, capture_output=True, text=True, ) report["proto"] = { "item_name_source": item_conversion, "mob_name_source": mob_conversion, "patch_result": result.stdout.strip().splitlines(), } report["pack"] = make_pack_dir(client_base, output / "pack", locale_dir, pack_tool) (output / "locale.cfg").write_text(LOCALE_CFG, encoding="ascii") (output / "locale_zh.cfg").write_text(LOCALE_CFG, encoding="ascii") # A local test checkout shares the original 40250 assets through symbolic links # (pack/ holds one per original pack); android/push-client.sh copies their targets. for name in ("sound", "BGM", "miles", "lib", "channel.inf"): source = client_base / name target = output / name if not source.exists(): continue if target.is_symlink(): if target.resolve() == source.resolve(): continue target.unlink() elif target.exists(): raise ValueError(f"refusing to replace existing client resource: {target}") target.symlink_to(source.resolve()) report_path = output.parent / "zh-locale-report.json" report_path.parent.mkdir(parents=True, exist_ok=True) report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") return report_path def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--english", type=Path, default=DEFAULT_EN) parser.add_argument("--chinese", type=Path, default=DEFAULT_ZH) parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT) parser.add_argument("--client-base", type=Path, default=REPO / "Client") parser.add_argument("--proto-tool", type=Path, default=tool("locale_proto_patch")) parser.add_argument("--pack-tool", type=Path, default=tool("eterpack_pack")) parser.add_argument("--name-tables", type=Path, default=DEFAULT_NAMES) args = parser.parse_args() report_path = build(args.english, args.chinese, args.output, args.client_base, args.proto_tool, args.name_tables, args.pack_tool) print(f"Built {args.output / 'pack/locale_zh.eix'} (.epk), registered in pack/Index") print(f"Report: {report_path}") if __name__ == "__main__": main()