40250 quests say gameforge.<quest>.<key>, loaded by game from share/locale/<locale>/translate.lua at start. build_zh_translate_lua.py lays the reviewed data/zh_40250/quest_translate_zh.tsv over the English file line by line (nested tables kept), re-wraps at the quest window width, and writes GBK with 0x5C trail bytes doubled. Deployed on 203; blacksmith dialog shows Chinese. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
175 lines
6.9 KiB
Python
175 lines
6.9 KiB
Python
#!/usr/bin/env python3
|
|
"""Build a Chinese translate.lua for the 40250 quest system.
|
|
|
|
The 40250 quests say `gameforge.<quest>.<key>`; game loads these strings from
|
|
share/locale/<locale>/translate.lua when it starts (questlua.cpp), so a new
|
|
translate.lua only needs a restart, not a quest recompile.
|
|
|
|
data/zh_40250/quest_translate_zh.tsv holds the reviewed translations, one
|
|
`<quest>.<key>\\t<text>` row per string, UTF-8. `[ENTER]` in a translation is a
|
|
deliberate line break; each line is re-wrapped to the quest window width
|
|
(CJK characters count double). A key without a row keeps its English text, and a
|
|
row fills every key whose English text equals the translated key's.
|
|
|
|
build_zh_translate_lua.py build <out translate.lua> GBK, for the server
|
|
build_zh_translate_lua.py todo <dir> [--quests a,b] [--chunk N]
|
|
untranslated unique strings
|
|
"""
|
|
|
|
import argparse
|
|
import re
|
|
import unicodedata
|
|
from pathlib import Path
|
|
|
|
|
|
REPO = Path(__file__).resolve().parents[1]
|
|
DEFAULT_EN = REPO.parent / "40250/Server Client TMP4/Server/metin2/server/share/locale/english/translate.lua"
|
|
DEFAULT_TABLE = REPO / "data/zh_40250/quest_translate_zh.tsv"
|
|
ROW = re.compile(r'^gameforge\.(\S+)\s*=\s*"(.*)"\s*$')
|
|
WIDTH = 50 # columns; the English lines are wrapped at about 50 characters
|
|
|
|
|
|
def english_rows(path: Path):
|
|
"""[(key, text)] in file order; the file is cp1252."""
|
|
rows = []
|
|
for line in path.read_bytes().decode("cp1252").splitlines():
|
|
match = ROW.match(line)
|
|
if match:
|
|
rows.append((match.group(1), match.group(2)))
|
|
return rows
|
|
|
|
|
|
def table_rows(path: Path):
|
|
rows = {}
|
|
if path.is_file():
|
|
for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
|
if not line or line.startswith("#"):
|
|
continue
|
|
key, sep, text = line.partition("\t")
|
|
if not sep:
|
|
raise ValueError(f"{path}:{number}: no tab")
|
|
rows[key] = text
|
|
return rows
|
|
|
|
|
|
def columns(char: str) -> int:
|
|
return 2 if unicodedata.east_asian_width(char) in "WF" else 1
|
|
|
|
|
|
def wrap(line: str, width: int = WIDTH):
|
|
"""Break a line at the width; CJK text breaks anywhere, ASCII words and %s stay whole."""
|
|
tokens = re.findall(r"%[-+ #0-9.]*[a-zA-Z]|[A-Za-z0-9_'.,!?:;()/-]+|\s+|.", line)
|
|
lines, current, used = [], "", 0
|
|
for token in tokens:
|
|
size = sum(columns(c) for c in token)
|
|
if used + size > width and current.strip():
|
|
lines.append(current.rstrip())
|
|
current, used = "", 0
|
|
if token.isspace():
|
|
continue
|
|
current += token
|
|
used += size
|
|
lines.append(current)
|
|
return lines
|
|
|
|
|
|
def layout(text: str) -> str:
|
|
return "[ENTER]".join(part for line in text.split("[ENTER]") for part in wrap(line))
|
|
|
|
|
|
# English text left untranslated, in characters GBK has no code for.
|
|
ASCII = str.maketrans({"\u2018": "'", "\u2019": "'", "\u00b4": "'", "\u201c": "'", "\u201d": "'",
|
|
"\u2013": "-", "\u2014": "-", "\u2026": "...", "\u00a0": " "})
|
|
|
|
|
|
def english_text(text: str) -> str:
|
|
text = text.translate(ASCII)
|
|
return "".join(c if c.isascii() else "?" for c in text)
|
|
|
|
|
|
def translations(english, table):
|
|
"""key -> Chinese text for every key, by key or by an equal English text."""
|
|
by_text = {}
|
|
for key, text in english:
|
|
if key in table:
|
|
by_text.setdefault(text, table[key])
|
|
return {key: table.get(key, by_text.get(text)) for key, text in english}
|
|
|
|
|
|
def build(en: Path, table_path: Path, output: Path):
|
|
english = english_rows(en)
|
|
table = table_rows(table_path)
|
|
unknown = sorted(set(table) - {key for key, _ in english})
|
|
if unknown:
|
|
raise ValueError(f"keys not in translate.lua: {unknown[:10]}")
|
|
chinese = translations(english, table)
|
|
lines = []
|
|
done = 0
|
|
# The English file line by line (it also creates the nested tables), with the values replaced.
|
|
for line in en.read_bytes().decode("cp1252").splitlines():
|
|
match = ROW.match(line)
|
|
if line.startswith("exportTestForCharset"):
|
|
lines.append('exportTestForCharset = "\u4e2d\u6587 "'.encode("gbk"))
|
|
continue
|
|
if not match:
|
|
lines.append(english_text(line).encode("ascii"))
|
|
continue
|
|
key, text = match.groups()
|
|
if chinese[key] is not None:
|
|
# Keep the English leading/trailing spaces: quests join some strings with names.
|
|
lead = text[: len(text) - len(text.lstrip(" "))]
|
|
trail = text[len(text.rstrip(" ")):]
|
|
body = layout(chinese[key].strip(" "))
|
|
if '"' in body or "\\" in body:
|
|
raise ValueError(f"{key}: no quotes or backslashes in a translation")
|
|
# A GBK trail byte can be 0x5C, which Lua reads as an escape; doubled it stays one byte.
|
|
value = lead.encode() + body.encode("gbk").replace(b"\\", b"\\\\") + trail.encode()
|
|
done += 1
|
|
else:
|
|
value = english_text(text).encode("ascii") # Lua source as in the English file
|
|
lines.append(f"gameforge.{key} = \"".encode("ascii") + value + b'"')
|
|
output.write_bytes(b"\n".join(lines) + b"\n")
|
|
return done, len(english)
|
|
|
|
|
|
def todo(en: Path, table_path: Path, out_dir: Path, quests=None, chunk=120):
|
|
english = english_rows(en)
|
|
chinese = translations(english, table_rows(table_path))
|
|
seen, pending = set(), []
|
|
for key, text in english:
|
|
if chinese[key] is None and text not in seen and (not quests or key.split(".")[0] in quests):
|
|
seen.add(text)
|
|
pending.append((key, text))
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
for old in out_dir.glob("todo_*.tsv"):
|
|
old.unlink()
|
|
for index in range(0, len(pending), chunk):
|
|
part = pending[index:index + chunk]
|
|
(out_dir / f"todo_{index // chunk:03d}.tsv").write_text(
|
|
"".join(f"{key}\t{text}\n" for key, text in part), encoding="utf-8")
|
|
return len(pending)
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
parser.add_argument("--english", type=Path, default=DEFAULT_EN)
|
|
parser.add_argument("--table", type=Path, default=DEFAULT_TABLE)
|
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
build_parser = sub.add_parser("build")
|
|
build_parser.add_argument("output", type=Path)
|
|
todo_parser = sub.add_parser("todo")
|
|
todo_parser.add_argument("out_dir", type=Path)
|
|
todo_parser.add_argument("--quests")
|
|
todo_parser.add_argument("--chunk", type=int, default=120)
|
|
args = parser.parse_args()
|
|
if args.command == "build":
|
|
done, total = build(args.english, args.table, args.output)
|
|
print(f"{args.output}: {done} of {total} strings in Chinese")
|
|
else:
|
|
quests = set(args.quests.split(",")) if args.quests else None
|
|
print(f"{todo(args.english, args.table, args.out_dir, quests, args.chunk)} strings to translate")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|