zh locale: translate all 40250 itemdesc descriptions, wrapped for the tooltip
itemdesc_translate_zh.tsv maps the English description/summary text to Chinese (1,788 entries, every itemdesc row covered). build_zh_locale.py replaces the cells and wraps Chinese into 26-byte lines joined by ' |', which 40250's Western SplitDescription turns into forced line breaks; the 50307 overlay row is wrapped the same way. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
3fcb27dd4d
commit
b7c215ee42
@@ -36,6 +36,13 @@ SKILL_COLUMNS = (2, 3, 4, 5, 6, 7, 8, 17, 20, 23)
|
||||
# vnum-keyed 40250 files whose rows an overlay (<stem>_zh_overlay.txt) replaces whole: items the server
|
||||
# gives a new use (tools/server/hunt-set) get a new icon and description.
|
||||
ROW_FILES = ("item_list.txt", "itemdesc.txt")
|
||||
# itemdesc.txt columns (0-based: description, summary) translated by their English text (itemdesc_translate_zh.tsv).
|
||||
ITEMDESC_COLUMNS = (2, 3)
|
||||
# The zh client keeps 40250's Western tooltip path (uitooltip.SplitDescription): lines break at spaces, at most
|
||||
# 35 bytes, and a token "|x" starts a new line. Chinese is wrapped here to the Eastern tooltip width instead
|
||||
# (DESC_DEFAULT_MAX_COLS = 26 bytes, a CJK character is 2) and each line after the first starts with "|".
|
||||
DESC_COLUMNS = 26
|
||||
BREAK_AFTER = ",。、;:!?)”"
|
||||
|
||||
|
||||
def keyed_lines(path: Path, encoding: str):
|
||||
@@ -76,13 +83,88 @@ def convert_names(source: Path, destination: Path, overlay: Path = None):
|
||||
return {"converted": converted, "overlaid": len(overlaid), "unrepresentable_vnums": unrepresentable}
|
||||
|
||||
|
||||
def overlay_rows(path: Path, overlay: Path):
|
||||
def wrap_lines(text: str, columns: int):
|
||||
atoms = re.findall(r"[\x21-\x7e]+|\s+|.", text.strip())
|
||||
lines, line, width, cut = [], [], 0, None
|
||||
for atom in atoms:
|
||||
size = len(atom.encode("gbk"))
|
||||
if line and width + size > columns and atom not in BREAK_AFTER:
|
||||
rest = []
|
||||
if cut is not None and cut >= len(line) // 2:
|
||||
rest, line = line[cut:], line[:cut]
|
||||
lines.append("".join(line).strip())
|
||||
while rest and rest[0].isspace():
|
||||
rest.pop(0)
|
||||
line, cut = rest, None
|
||||
width = sum(len(a.encode("gbk")) for a in line)
|
||||
if atom.isspace() and not line:
|
||||
continue
|
||||
line.append(atom)
|
||||
width += size
|
||||
if atom in BREAK_AFTER:
|
||||
cut = len(line)
|
||||
if line:
|
||||
lines.append("".join(line).strip())
|
||||
return lines
|
||||
|
||||
|
||||
def wrap_description(text: str) -> str:
|
||||
"""Chinese text as SplitDescription lines of at most DESC_COLUMNS bytes. ASCII runs (numbers, words) are not
|
||||
split; a line ends after punctuation when one is in its second half, and punctuation never starts a line
|
||||
(it may hang one character past the width, which the 35-byte Western limit still keeps on the line).
|
||||
A last line of one or two characters is avoided by narrowing the lines when that needs no extra line."""
|
||||
if all(ord(c) < 128 for c in text) or "|" in text:
|
||||
return text
|
||||
lines = wrap_lines(text, DESC_COLUMNS)
|
||||
for columns in range(DESC_COLUMNS - 2, DESC_COLUMNS - 9, -2):
|
||||
if len(lines) < 2 or len(lines[-1].encode("gbk")) > 6:
|
||||
break
|
||||
narrower = wrap_lines(text, columns)
|
||||
if len(narrower) == len(lines):
|
||||
lines = narrower
|
||||
return " |".join(lines)
|
||||
|
||||
|
||||
def translate_itemdesc(path: Path, table: Path):
|
||||
"""itemdesc.txt with its description and summary columns replaced from the English→Chinese table (UTF-8,
|
||||
English<TAB>Chinese) and wrapped; untranslated cells keep the English text."""
|
||||
chinese = {}
|
||||
for line in table.read_text(encoding="utf-8").splitlines():
|
||||
if line and not line.startswith("#"):
|
||||
english, text = line.split("\t")
|
||||
chinese[english] = wrap_description(text)
|
||||
rows, translated, english_rows = [], 0, []
|
||||
for line in path.read_bytes().decode("cp1252").split("\n"):
|
||||
parts = line.split("\t")
|
||||
if parts[0].strip().isdigit():
|
||||
left = False
|
||||
for column in ITEMDESC_COLUMNS:
|
||||
text = parts[column].strip() if column < len(parts) else ""
|
||||
if not text:
|
||||
continue
|
||||
if text in chinese:
|
||||
parts[column] = chinese[text] + ("\r" if parts[column].endswith("\r") else "")
|
||||
translated += 1
|
||||
else:
|
||||
left = True
|
||||
if left:
|
||||
english_rows.append(parts[0].strip())
|
||||
rows.append("\t".join(parts))
|
||||
path.write_bytes("\n".join(rows).encode("gbk"))
|
||||
return {"translated_cells": translated, "english_rows": english_rows}
|
||||
|
||||
|
||||
def overlay_rows(path: Path, overlay: Path, wrap_columns=()):
|
||||
"""Replace the rows of a vnum-keyed file by the overlay's (UTF-8, whole rows, no header); other rows keep
|
||||
their bytes (the English files are cp1252)."""
|
||||
their bytes (the English files are cp1252). wrap_columns (0-based) are wrapped like translated text."""
|
||||
rows = {}
|
||||
for line in overlay.read_text(encoding="utf-8").splitlines():
|
||||
if line and not line.startswith("#"):
|
||||
rows[line.split("\t", 1)[0]] = line.encode("gbk")
|
||||
parts = line.split("\t")
|
||||
for column in wrap_columns:
|
||||
if column < len(parts):
|
||||
parts[column] = wrap_description(parts[column])
|
||||
rows[parts[0]] = "\t".join(parts).encode("gbk")
|
||||
lines = path.read_bytes().split(b"\n")
|
||||
replaced = []
|
||||
for index, line in enumerate(lines):
|
||||
@@ -236,10 +318,15 @@ def build(en: Path, zh: Path, output: Path, client_base: Path,
|
||||
en / "skilldesc.txt", zh / "skilldesc.txt", locale_dir / "skilldesc.txt",
|
||||
names_dir / "skilldesc_zh_overlay.txt" if names_dir else None)
|
||||
|
||||
table = names_dir / "itemdesc_translate_zh.tsv" if names_dir else None
|
||||
if table and table.is_file():
|
||||
report["files"]["itemdesc.txt translation"] = translate_itemdesc(locale_dir / "itemdesc.txt", table)
|
||||
|
||||
for filename in ROW_FILES:
|
||||
overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None
|
||||
if overlay and overlay.is_file():
|
||||
report["files"][filename] = overlay_rows(locale_dir / filename, overlay)
|
||||
wrap = ITEMDESC_COLUMNS if filename == "itemdesc.txt" else ()
|
||||
report["files"][filename] = overlay_rows(locale_dir / filename, overlay, wrap)
|
||||
|
||||
for folder in IMAGE_DIRS:
|
||||
english_files = {path.name for path in (en / folder).iterdir()}
|
||||
|
||||
@@ -13,6 +13,29 @@ MODULE = importlib.util.module_from_spec(SPEC)
|
||||
SPEC.loader.exec_module(MODULE)
|
||||
|
||||
|
||||
def split_description(desc, limit):
|
||||
"""uitooltip.SplitDescription (40250 root), which runs on the encoded bytes under Python 2."""
|
||||
line_tokens, line_len, lines = [], 0, []
|
||||
for token in desc.split():
|
||||
if b"|" in token:
|
||||
sep_pos = token.find(b"|")
|
||||
line_tokens.append(token[:sep_pos])
|
||||
lines.append(b" ".join(line_tokens))
|
||||
line_len = len(token) - (sep_pos + 1)
|
||||
line_tokens = [token[sep_pos + 1:]]
|
||||
else:
|
||||
line_len += len(token)
|
||||
if len(line_tokens) + line_len > limit:
|
||||
lines.append(b" ".join(line_tokens))
|
||||
line_len = len(token)
|
||||
line_tokens = [token]
|
||||
else:
|
||||
line_tokens.append(token)
|
||||
if line_tokens:
|
||||
lines.append(b" ".join(line_tokens))
|
||||
return lines
|
||||
|
||||
|
||||
class LocaleBuildTest(unittest.TestCase):
|
||||
def test_preserves_40250_keys_formats_and_proto(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
@@ -95,6 +118,31 @@ class LocaleBuildTest(unittest.TestCase):
|
||||
with self.assertRaises(ValueError):
|
||||
MODULE.overlay_rows(root / "itemdesc.txt", root / "overlay.txt")
|
||||
|
||||
def test_itemdesc_translation_wraps_for_split_description(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
(root / "itemdesc.txt").write_bytes(
|
||||
b"27001\tRed Potion (S)\tRestores HP.\t\r\n"
|
||||
b"27002\tRed Potion (M)\tRestores 1,000 HP. Item is tradeable.\tPotion production\r\n"
|
||||
b"99\tNew\tCaf\xe9.\t\r\n")
|
||||
(root / "table.tsv").write_text(
|
||||
"# English\t中文\nRestores HP.\t恢复生命值。\n"
|
||||
"Restores 1,000 HP. Item is tradeable.\t立即恢复1,000点生命值,此物品可交易,冷却时间结束后可再次使用。\n"
|
||||
"Potion production\t药水制作\n")
|
||||
result = MODULE.translate_itemdesc(root / "itemdesc.txt", root / "table.tsv")
|
||||
rows = (root / "itemdesc.txt").read_bytes().decode("gbk").split("\r\n")
|
||||
self.assertEqual(rows[0], "27001\tRed Potion (S)\t恢复生命值。\t")
|
||||
wrapped = rows[1].split("\t")[2]
|
||||
self.assertEqual(wrapped, "立即恢复1,000点生命值, |此物品可交易, |冷却时间结束后可再次使用。")
|
||||
self.assertEqual(rows[1].split("\t")[3], "药水制作")
|
||||
self.assertEqual(rows[2], "99\tNew\tCafé.\t")
|
||||
self.assertEqual(result, {"translated_cells": 3, "english_rows": ["99"]})
|
||||
# uitooltip.SplitDescription (40250, Western path, 35 columns) on the GBK bytes gives these lines.
|
||||
lines = split_description(wrapped.encode("gbk"), 35)
|
||||
self.assertEqual([line.strip().decode("gbk") for line in lines],
|
||||
["立即恢复1,000点生命值,", "此物品可交易,", "冷却时间结束后可再次使用。"])
|
||||
self.assertEqual(MODULE.wrap_description("Plain English text"), "Plain English text")
|
||||
|
||||
def test_skilldesc_keeps_format_arguments(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
|
||||
Reference in New Issue
Block a user