zh locale: translate all 40250 itemdesc descriptions, wrapped for the tooltip
itemdesc_translate_zh.tsv maps the English description/summary text to Chinese (1,788 entries, every itemdesc row covered). build_zh_locale.py replaces the cells and wraps Chinese into 26-byte lines joined by ' |', which 40250's Western SplitDescription turns into forced line breaks; the 50307 overlay row is wrapped the same way. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
3fcb27dd4d
commit
b7c215ee42
@@ -44,3 +44,16 @@
|
|||||||
`item_list_zh_overlay.txt`、`itemdesc_zh_overlay.txt` 每行是目标文件的一整行(UTF-8,第一列 VNUM),
|
`item_list_zh_overlay.txt`、`itemdesc_zh_overlay.txt` 每行是目标文件的一整行(UTF-8,第一列 VNUM),
|
||||||
`build_zh_locale.py` 用它替换同 VNUM 的行(`ROW_FILES`),其余行字节不变;VNUM 不存在就报错。
|
`build_zh_locale.py` 用它替换同 VNUM 的行(`ROW_FILES`),其余行字节不变;VNUM 不存在就报错。
|
||||||
用于服务端改了用途的物品,例如 50307 小狩猎套装礼包(`tools/server/hunt-set`)换成礼盒图标和中文说明。
|
用于服务端改了用途的物品,例如 50307 小狩猎套装礼包(`tools/server/hunt-set`)换成礼盒图标和中文说明。
|
||||||
|
|
||||||
|
## 物品说明(itemdesc_translate_zh.tsv)
|
||||||
|
|
||||||
|
40250 `itemdesc.txt` 的说明列(第 3 列)和摘要列(第 4 列,“Potion production”“Research”)按英文原文整句翻译,
|
||||||
|
一共 1,788 条,放在 `itemdesc_translate_zh.tsv`(英文、制表符、中文,UTF-8)。英文原文完全相同的物品共用同一条译文。
|
||||||
|
第 2 列物品名不改:客户端显示的名称来自 `item_proto`(见上文名称快照)。
|
||||||
|
|
||||||
|
- 客户端的 tooltip 走的是 40250 的西文路径(`uitooltip.SplitDescription`):只在空格处断行,每行最多 35 字节,
|
||||||
|
遇到 `|x` 这种词会另起一行。构建时会把中文折成东方路径的宽度,也就是每行 26 字节(一个汉字算 2 字节),再用 ` |` 连接各行。
|
||||||
|
- 断行优先放在标点后面,标点不放在行首;英文单词和数字不会被拆开。
|
||||||
|
- 如果最后一行只剩一两个字,会把前面的行收窄一些重新折。
|
||||||
|
- 表里没有的格子保留英文,`zh-locale-report.json` 里的 `itemdesc.txt translation.english_rows` 会列出这些 VNUM。
|
||||||
|
- `itemdesc_zh_overlay.txt` 的整行覆盖在翻译之后执行,它的说明列也用同样的方式折行。
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -36,6 +36,13 @@ SKILL_COLUMNS = (2, 3, 4, 5, 6, 7, 8, 17, 20, 23)
|
|||||||
# vnum-keyed 40250 files whose rows an overlay (<stem>_zh_overlay.txt) replaces whole: items the server
|
# vnum-keyed 40250 files whose rows an overlay (<stem>_zh_overlay.txt) replaces whole: items the server
|
||||||
# gives a new use (tools/server/hunt-set) get a new icon and description.
|
# gives a new use (tools/server/hunt-set) get a new icon and description.
|
||||||
ROW_FILES = ("item_list.txt", "itemdesc.txt")
|
ROW_FILES = ("item_list.txt", "itemdesc.txt")
|
||||||
|
# itemdesc.txt columns (0-based: description, summary) translated by their English text (itemdesc_translate_zh.tsv).
|
||||||
|
ITEMDESC_COLUMNS = (2, 3)
|
||||||
|
# The zh client keeps 40250's Western tooltip path (uitooltip.SplitDescription): lines break at spaces, at most
|
||||||
|
# 35 bytes, and a token "|x" starts a new line. Chinese is wrapped here to the Eastern tooltip width instead
|
||||||
|
# (DESC_DEFAULT_MAX_COLS = 26 bytes, a CJK character is 2) and each line after the first starts with "|".
|
||||||
|
DESC_COLUMNS = 26
|
||||||
|
BREAK_AFTER = ",。、;:!?)”"
|
||||||
|
|
||||||
|
|
||||||
def keyed_lines(path: Path, encoding: str):
|
def keyed_lines(path: Path, encoding: str):
|
||||||
@@ -76,13 +83,88 @@ def convert_names(source: Path, destination: Path, overlay: Path = None):
|
|||||||
return {"converted": converted, "overlaid": len(overlaid), "unrepresentable_vnums": unrepresentable}
|
return {"converted": converted, "overlaid": len(overlaid), "unrepresentable_vnums": unrepresentable}
|
||||||
|
|
||||||
|
|
||||||
def overlay_rows(path: Path, overlay: Path):
|
def wrap_lines(text: str, columns: int):
|
||||||
|
atoms = re.findall(r"[\x21-\x7e]+|\s+|.", text.strip())
|
||||||
|
lines, line, width, cut = [], [], 0, None
|
||||||
|
for atom in atoms:
|
||||||
|
size = len(atom.encode("gbk"))
|
||||||
|
if line and width + size > columns and atom not in BREAK_AFTER:
|
||||||
|
rest = []
|
||||||
|
if cut is not None and cut >= len(line) // 2:
|
||||||
|
rest, line = line[cut:], line[:cut]
|
||||||
|
lines.append("".join(line).strip())
|
||||||
|
while rest and rest[0].isspace():
|
||||||
|
rest.pop(0)
|
||||||
|
line, cut = rest, None
|
||||||
|
width = sum(len(a.encode("gbk")) for a in line)
|
||||||
|
if atom.isspace() and not line:
|
||||||
|
continue
|
||||||
|
line.append(atom)
|
||||||
|
width += size
|
||||||
|
if atom in BREAK_AFTER:
|
||||||
|
cut = len(line)
|
||||||
|
if line:
|
||||||
|
lines.append("".join(line).strip())
|
||||||
|
return lines
|
||||||
|
|
||||||
|
|
||||||
|
def wrap_description(text: str) -> str:
|
||||||
|
"""Chinese text as SplitDescription lines of at most DESC_COLUMNS bytes. ASCII runs (numbers, words) are not
|
||||||
|
split; a line ends after punctuation when one is in its second half, and punctuation never starts a line
|
||||||
|
(it may hang one character past the width, which the 35-byte Western limit still keeps on the line).
|
||||||
|
A last line of one or two characters is avoided by narrowing the lines when that needs no extra line."""
|
||||||
|
if all(ord(c) < 128 for c in text) or "|" in text:
|
||||||
|
return text
|
||||||
|
lines = wrap_lines(text, DESC_COLUMNS)
|
||||||
|
for columns in range(DESC_COLUMNS - 2, DESC_COLUMNS - 9, -2):
|
||||||
|
if len(lines) < 2 or len(lines[-1].encode("gbk")) > 6:
|
||||||
|
break
|
||||||
|
narrower = wrap_lines(text, columns)
|
||||||
|
if len(narrower) == len(lines):
|
||||||
|
lines = narrower
|
||||||
|
return " |".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def translate_itemdesc(path: Path, table: Path):
|
||||||
|
"""itemdesc.txt with its description and summary columns replaced from the English→Chinese table (UTF-8,
|
||||||
|
English<TAB>Chinese) and wrapped; untranslated cells keep the English text."""
|
||||||
|
chinese = {}
|
||||||
|
for line in table.read_text(encoding="utf-8").splitlines():
|
||||||
|
if line and not line.startswith("#"):
|
||||||
|
english, text = line.split("\t")
|
||||||
|
chinese[english] = wrap_description(text)
|
||||||
|
rows, translated, english_rows = [], 0, []
|
||||||
|
for line in path.read_bytes().decode("cp1252").split("\n"):
|
||||||
|
parts = line.split("\t")
|
||||||
|
if parts[0].strip().isdigit():
|
||||||
|
left = False
|
||||||
|
for column in ITEMDESC_COLUMNS:
|
||||||
|
text = parts[column].strip() if column < len(parts) else ""
|
||||||
|
if not text:
|
||||||
|
continue
|
||||||
|
if text in chinese:
|
||||||
|
parts[column] = chinese[text] + ("\r" if parts[column].endswith("\r") else "")
|
||||||
|
translated += 1
|
||||||
|
else:
|
||||||
|
left = True
|
||||||
|
if left:
|
||||||
|
english_rows.append(parts[0].strip())
|
||||||
|
rows.append("\t".join(parts))
|
||||||
|
path.write_bytes("\n".join(rows).encode("gbk"))
|
||||||
|
return {"translated_cells": translated, "english_rows": english_rows}
|
||||||
|
|
||||||
|
|
||||||
|
def overlay_rows(path: Path, overlay: Path, wrap_columns=()):
|
||||||
"""Replace the rows of a vnum-keyed file by the overlay's (UTF-8, whole rows, no header); other rows keep
|
"""Replace the rows of a vnum-keyed file by the overlay's (UTF-8, whole rows, no header); other rows keep
|
||||||
their bytes (the English files are cp1252)."""
|
their bytes (the English files are cp1252). wrap_columns (0-based) are wrapped like translated text."""
|
||||||
rows = {}
|
rows = {}
|
||||||
for line in overlay.read_text(encoding="utf-8").splitlines():
|
for line in overlay.read_text(encoding="utf-8").splitlines():
|
||||||
if line and not line.startswith("#"):
|
if line and not line.startswith("#"):
|
||||||
rows[line.split("\t", 1)[0]] = line.encode("gbk")
|
parts = line.split("\t")
|
||||||
|
for column in wrap_columns:
|
||||||
|
if column < len(parts):
|
||||||
|
parts[column] = wrap_description(parts[column])
|
||||||
|
rows[parts[0]] = "\t".join(parts).encode("gbk")
|
||||||
lines = path.read_bytes().split(b"\n")
|
lines = path.read_bytes().split(b"\n")
|
||||||
replaced = []
|
replaced = []
|
||||||
for index, line in enumerate(lines):
|
for index, line in enumerate(lines):
|
||||||
@@ -236,10 +318,15 @@ def build(en: Path, zh: Path, output: Path, client_base: Path,
|
|||||||
en / "skilldesc.txt", zh / "skilldesc.txt", locale_dir / "skilldesc.txt",
|
en / "skilldesc.txt", zh / "skilldesc.txt", locale_dir / "skilldesc.txt",
|
||||||
names_dir / "skilldesc_zh_overlay.txt" if names_dir else None)
|
names_dir / "skilldesc_zh_overlay.txt" if names_dir else None)
|
||||||
|
|
||||||
|
table = names_dir / "itemdesc_translate_zh.tsv" if names_dir else None
|
||||||
|
if table and table.is_file():
|
||||||
|
report["files"]["itemdesc.txt translation"] = translate_itemdesc(locale_dir / "itemdesc.txt", table)
|
||||||
|
|
||||||
for filename in ROW_FILES:
|
for filename in ROW_FILES:
|
||||||
overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None
|
overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None
|
||||||
if overlay and overlay.is_file():
|
if overlay and overlay.is_file():
|
||||||
report["files"][filename] = overlay_rows(locale_dir / filename, overlay)
|
wrap = ITEMDESC_COLUMNS if filename == "itemdesc.txt" else ()
|
||||||
|
report["files"][filename] = overlay_rows(locale_dir / filename, overlay, wrap)
|
||||||
|
|
||||||
for folder in IMAGE_DIRS:
|
for folder in IMAGE_DIRS:
|
||||||
english_files = {path.name for path in (en / folder).iterdir()}
|
english_files = {path.name for path in (en / folder).iterdir()}
|
||||||
|
|||||||
@@ -13,6 +13,29 @@ MODULE = importlib.util.module_from_spec(SPEC)
|
|||||||
SPEC.loader.exec_module(MODULE)
|
SPEC.loader.exec_module(MODULE)
|
||||||
|
|
||||||
|
|
||||||
|
def split_description(desc, limit):
|
||||||
|
"""uitooltip.SplitDescription (40250 root), which runs on the encoded bytes under Python 2."""
|
||||||
|
line_tokens, line_len, lines = [], 0, []
|
||||||
|
for token in desc.split():
|
||||||
|
if b"|" in token:
|
||||||
|
sep_pos = token.find(b"|")
|
||||||
|
line_tokens.append(token[:sep_pos])
|
||||||
|
lines.append(b" ".join(line_tokens))
|
||||||
|
line_len = len(token) - (sep_pos + 1)
|
||||||
|
line_tokens = [token[sep_pos + 1:]]
|
||||||
|
else:
|
||||||
|
line_len += len(token)
|
||||||
|
if len(line_tokens) + line_len > limit:
|
||||||
|
lines.append(b" ".join(line_tokens))
|
||||||
|
line_len = len(token)
|
||||||
|
line_tokens = [token]
|
||||||
|
else:
|
||||||
|
line_tokens.append(token)
|
||||||
|
if line_tokens:
|
||||||
|
lines.append(b" ".join(line_tokens))
|
||||||
|
return lines
|
||||||
|
|
||||||
|
|
||||||
class LocaleBuildTest(unittest.TestCase):
|
class LocaleBuildTest(unittest.TestCase):
|
||||||
def test_preserves_40250_keys_formats_and_proto(self):
|
def test_preserves_40250_keys_formats_and_proto(self):
|
||||||
with tempfile.TemporaryDirectory() as directory:
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
@@ -95,6 +118,31 @@ class LocaleBuildTest(unittest.TestCase):
|
|||||||
with self.assertRaises(ValueError):
|
with self.assertRaises(ValueError):
|
||||||
MODULE.overlay_rows(root / "itemdesc.txt", root / "overlay.txt")
|
MODULE.overlay_rows(root / "itemdesc.txt", root / "overlay.txt")
|
||||||
|
|
||||||
|
def test_itemdesc_translation_wraps_for_split_description(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
root = Path(directory)
|
||||||
|
(root / "itemdesc.txt").write_bytes(
|
||||||
|
b"27001\tRed Potion (S)\tRestores HP.\t\r\n"
|
||||||
|
b"27002\tRed Potion (M)\tRestores 1,000 HP. Item is tradeable.\tPotion production\r\n"
|
||||||
|
b"99\tNew\tCaf\xe9.\t\r\n")
|
||||||
|
(root / "table.tsv").write_text(
|
||||||
|
"# English\t中文\nRestores HP.\t恢复生命值。\n"
|
||||||
|
"Restores 1,000 HP. Item is tradeable.\t立即恢复1,000点生命值,此物品可交易,冷却时间结束后可再次使用。\n"
|
||||||
|
"Potion production\t药水制作\n")
|
||||||
|
result = MODULE.translate_itemdesc(root / "itemdesc.txt", root / "table.tsv")
|
||||||
|
rows = (root / "itemdesc.txt").read_bytes().decode("gbk").split("\r\n")
|
||||||
|
self.assertEqual(rows[0], "27001\tRed Potion (S)\t恢复生命值。\t")
|
||||||
|
wrapped = rows[1].split("\t")[2]
|
||||||
|
self.assertEqual(wrapped, "立即恢复1,000点生命值, |此物品可交易, |冷却时间结束后可再次使用。")
|
||||||
|
self.assertEqual(rows[1].split("\t")[3], "药水制作")
|
||||||
|
self.assertEqual(rows[2], "99\tNew\tCafé.\t")
|
||||||
|
self.assertEqual(result, {"translated_cells": 3, "english_rows": ["99"]})
|
||||||
|
# uitooltip.SplitDescription (40250, Western path, 35 columns) on the GBK bytes gives these lines.
|
||||||
|
lines = split_description(wrapped.encode("gbk"), 35)
|
||||||
|
self.assertEqual([line.strip().decode("gbk") for line in lines],
|
||||||
|
["立即恢复1,000点生命值,", "此物品可交易,", "冷却时间结束后可再次使用。"])
|
||||||
|
self.assertEqual(MODULE.wrap_description("Plain English text"), "Plain English text")
|
||||||
|
|
||||||
def test_skilldesc_keeps_format_arguments(self):
|
def test_skilldesc_keeps_format_arguments(self):
|
||||||
with tempfile.TemporaryDirectory() as directory:
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
root = Path(directory)
|
root = Path(directory)
|
||||||
|
|||||||
Reference in New Issue
Block a user