zh locale: translate all 40250 itemdesc descriptions, wrapped for the tooltip

itemdesc_translate_zh.tsv maps the English description/summary text to Chinese (1,788 entries,
every itemdesc row covered). build_zh_locale.py replaces the cells and wraps Chinese into
26-byte lines joined by ' |', which 40250's Western SplitDescription turns into forced line
breaks; the 50307 overlay row is wrapped the same way.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-30 11:17:21 +09:00
co-authored by Claude Opus 5.5
parent 3fcb27dd4d
commit b7c215ee42
4 changed files with 1941 additions and 4 deletions
+13
View File
@@ -44,3 +44,16 @@
`item_list_zh_overlay.txt`、`itemdesc_zh_overlay.txt` 每行是目标文件的一整行(UTF-8,第一列 VNUM),
`build_zh_locale.py` 用它替换同 VNUM 的行(`ROW_FILES`),其余行字节不变;VNUM 不存在就报错。
用于服务端改了用途的物品,例如 50307 小狩猎套装礼包(`tools/server/hunt-set`)换成礼盒图标和中文说明。
## 物品说明(itemdesc_translate_zh.tsv)
40250 `itemdesc.txt` 的说明列(第 3 列)和摘要列(第 4 列,“Potion production”“Research”)按英文原文整句翻译,
一共 1,788 条,放在 `itemdesc_translate_zh.tsv`(英文、制表符、中文,UTF-8)。英文原文完全相同的物品共用同一条译文。
第 2 列物品名不改:客户端显示的名称来自 `item_proto`(见上文名称快照)。
- 客户端的 tooltip 走的是 40250 的西文路径(`uitooltip.SplitDescription`):只在空格处断行,每行最多 35 字节,
遇到 `|x` 这种词会另起一行。构建时会把中文折成东方路径的宽度,也就是每行 26 字节(一个汉字算 2 字节),再用 ` |` 连接各行。
- 断行优先放在标点后面,标点不放在行首;英文单词和数字不会被拆开。
- 如果最后一行只剩一两个字,会把前面的行收窄一些重新折。
- 表里没有的格子保留英文,`zh-locale-report.json` 里的 `itemdesc.txt translation.english_rows` 会列出这些 VNUM。
- `itemdesc_zh_overlay.txt` 的整行覆盖在翻译之后执行,它的说明列也用同样的方式折行。
File diff suppressed because it is too large Load Diff
+91 -4
View File
@@ -36,6 +36,13 @@ SKILL_COLUMNS = (2, 3, 4, 5, 6, 7, 8, 17, 20, 23)
# vnum-keyed 40250 files whose rows an overlay (<stem>_zh_overlay.txt) replaces whole: items the server
# gives a new use (tools/server/hunt-set) get a new icon and description.
ROW_FILES = ("item_list.txt", "itemdesc.txt")
# itemdesc.txt columns (0-based: description, summary) translated by their English text (itemdesc_translate_zh.tsv).
ITEMDESC_COLUMNS = (2, 3)
# The zh client keeps 40250's Western tooltip path (uitooltip.SplitDescription): lines break at spaces, at most
# 35 bytes, and a token "|x" starts a new line. Chinese is wrapped here to the Eastern tooltip width instead
# (DESC_DEFAULT_MAX_COLS = 26 bytes, a CJK character is 2) and each line after the first starts with "|".
DESC_COLUMNS = 26
BREAK_AFTER = ",。、;:!?)”"
def keyed_lines(path: Path, encoding: str):
@@ -76,13 +83,88 @@ def convert_names(source: Path, destination: Path, overlay: Path = None):
return {"converted": converted, "overlaid": len(overlaid), "unrepresentable_vnums": unrepresentable}
def overlay_rows(path: Path, overlay: Path):
def wrap_lines(text: str, columns: int):
atoms = re.findall(r"[\x21-\x7e]+|\s+|.", text.strip())
lines, line, width, cut = [], [], 0, None
for atom in atoms:
size = len(atom.encode("gbk"))
if line and width + size > columns and atom not in BREAK_AFTER:
rest = []
if cut is not None and cut >= len(line) // 2:
rest, line = line[cut:], line[:cut]
lines.append("".join(line).strip())
while rest and rest[0].isspace():
rest.pop(0)
line, cut = rest, None
width = sum(len(a.encode("gbk")) for a in line)
if atom.isspace() and not line:
continue
line.append(atom)
width += size
if atom in BREAK_AFTER:
cut = len(line)
if line:
lines.append("".join(line).strip())
return lines
def wrap_description(text: str) -> str:
"""Chinese text as SplitDescription lines of at most DESC_COLUMNS bytes. ASCII runs (numbers, words) are not
split; a line ends after punctuation when one is in its second half, and punctuation never starts a line
(it may hang one character past the width, which the 35-byte Western limit still keeps on the line).
A last line of one or two characters is avoided by narrowing the lines when that needs no extra line."""
if all(ord(c) < 128 for c in text) or "|" in text:
return text
lines = wrap_lines(text, DESC_COLUMNS)
for columns in range(DESC_COLUMNS - 2, DESC_COLUMNS - 9, -2):
if len(lines) < 2 or len(lines[-1].encode("gbk")) > 6:
break
narrower = wrap_lines(text, columns)
if len(narrower) == len(lines):
lines = narrower
return " |".join(lines)
def translate_itemdesc(path: Path, table: Path):
"""itemdesc.txt with its description and summary columns replaced from the English→Chinese table (UTF-8,
English<TAB>Chinese) and wrapped; untranslated cells keep the English text."""
chinese = {}
for line in table.read_text(encoding="utf-8").splitlines():
if line and not line.startswith("#"):
english, text = line.split("\t")
chinese[english] = wrap_description(text)
rows, translated, english_rows = [], 0, []
for line in path.read_bytes().decode("cp1252").split("\n"):
parts = line.split("\t")
if parts[0].strip().isdigit():
left = False
for column in ITEMDESC_COLUMNS:
text = parts[column].strip() if column < len(parts) else ""
if not text:
continue
if text in chinese:
parts[column] = chinese[text] + ("\r" if parts[column].endswith("\r") else "")
translated += 1
else:
left = True
if left:
english_rows.append(parts[0].strip())
rows.append("\t".join(parts))
path.write_bytes("\n".join(rows).encode("gbk"))
return {"translated_cells": translated, "english_rows": english_rows}
def overlay_rows(path: Path, overlay: Path, wrap_columns=()):
"""Replace the rows of a vnum-keyed file by the overlay's (UTF-8, whole rows, no header); other rows keep
their bytes (the English files are cp1252)."""
their bytes (the English files are cp1252). wrap_columns (0-based) are wrapped like translated text."""
rows = {}
for line in overlay.read_text(encoding="utf-8").splitlines():
if line and not line.startswith("#"):
rows[line.split("\t", 1)[0]] = line.encode("gbk")
parts = line.split("\t")
for column in wrap_columns:
if column < len(parts):
parts[column] = wrap_description(parts[column])
rows[parts[0]] = "\t".join(parts).encode("gbk")
lines = path.read_bytes().split(b"\n")
replaced = []
for index, line in enumerate(lines):
@@ -236,10 +318,15 @@ def build(en: Path, zh: Path, output: Path, client_base: Path,
en / "skilldesc.txt", zh / "skilldesc.txt", locale_dir / "skilldesc.txt",
names_dir / "skilldesc_zh_overlay.txt" if names_dir else None)
table = names_dir / "itemdesc_translate_zh.tsv" if names_dir else None
if table and table.is_file():
report["files"]["itemdesc.txt translation"] = translate_itemdesc(locale_dir / "itemdesc.txt", table)
for filename in ROW_FILES:
overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None
if overlay and overlay.is_file():
report["files"][filename] = overlay_rows(locale_dir / filename, overlay)
wrap = ITEMDESC_COLUMNS if filename == "itemdesc.txt" else ()
report["files"][filename] = overlay_rows(locale_dir / filename, overlay, wrap)
for folder in IMAGE_DIRS:
english_files = {path.name for path in (en / folder).iterdir()}
+48
View File
@@ -13,6 +13,29 @@ MODULE = importlib.util.module_from_spec(SPEC)
SPEC.loader.exec_module(MODULE)
def split_description(desc, limit):
"""uitooltip.SplitDescription (40250 root), which runs on the encoded bytes under Python 2."""
line_tokens, line_len, lines = [], 0, []
for token in desc.split():
if b"|" in token:
sep_pos = token.find(b"|")
line_tokens.append(token[:sep_pos])
lines.append(b" ".join(line_tokens))
line_len = len(token) - (sep_pos + 1)
line_tokens = [token[sep_pos + 1:]]
else:
line_len += len(token)
if len(line_tokens) + line_len > limit:
lines.append(b" ".join(line_tokens))
line_len = len(token)
line_tokens = [token]
else:
line_tokens.append(token)
if line_tokens:
lines.append(b" ".join(line_tokens))
return lines
class LocaleBuildTest(unittest.TestCase):
def test_preserves_40250_keys_formats_and_proto(self):
with tempfile.TemporaryDirectory() as directory:
@@ -95,6 +118,31 @@ class LocaleBuildTest(unittest.TestCase):
with self.assertRaises(ValueError):
MODULE.overlay_rows(root / "itemdesc.txt", root / "overlay.txt")
def test_itemdesc_translation_wraps_for_split_description(self):
with tempfile.TemporaryDirectory() as directory:
root = Path(directory)
(root / "itemdesc.txt").write_bytes(
b"27001\tRed Potion (S)\tRestores HP.\t\r\n"
b"27002\tRed Potion (M)\tRestores 1,000 HP. Item is tradeable.\tPotion production\r\n"
b"99\tNew\tCaf\xe9.\t\r\n")
(root / "table.tsv").write_text(
"# English\t中文\nRestores HP.\t恢复生命值。\n"
"Restores 1,000 HP. Item is tradeable.\t立即恢复1,000点生命值,此物品可交易,冷却时间结束后可再次使用。\n"
"Potion production\t药水制作\n")
result = MODULE.translate_itemdesc(root / "itemdesc.txt", root / "table.tsv")
rows = (root / "itemdesc.txt").read_bytes().decode("gbk").split("\r\n")
self.assertEqual(rows[0], "27001\tRed Potion (S)\t恢复生命值。\t")
wrapped = rows[1].split("\t")[2]
self.assertEqual(wrapped, "立即恢复1,000点生命值, |此物品可交易, |冷却时间结束后可再次使用。")
self.assertEqual(rows[1].split("\t")[3], "药水制作")
self.assertEqual(rows[2], "99\tNew\tCafé.\t")
self.assertEqual(result, {"translated_cells": 3, "english_rows": ["99"]})
# uitooltip.SplitDescription (40250, Western path, 35 columns) on the GBK bytes gives these lines.
lines = split_description(wrapped.encode("gbk"), 35)
self.assertEqual([line.strip().decode("gbk") for line in lines],
["立即恢复1,000点生命值,", "此物品可交易,", "冷却时间结束后可再次使用。"])
self.assertEqual(MODULE.wrap_description("Plain English text"), "Plain English text")
def test_skilldesc_keeps_format_arguments(self):
with tempfile.TemporaryDirectory() as directory:
root = Path(directory)