zh locale: translate all 40250 itemdesc descriptions, wrapped for the tooltip

itemdesc_translate_zh.tsv maps the English description/summary text to Chinese (1,788 entries,
every itemdesc row covered). build_zh_locale.py replaces the cells and wraps Chinese into
26-byte lines joined by ' |', which 40250's Western SplitDescription turns into forced line
breaks; the 50307 overlay row is wrapped the same way.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-30 11:17:21 +09:00
co-authored by Claude Opus 5.5
parent 3fcb27dd4d
commit b7c215ee42
4 changed files with 1941 additions and 4 deletions
+91 -4
View File
@@ -36,6 +36,13 @@ SKILL_COLUMNS = (2, 3, 4, 5, 6, 7, 8, 17, 20, 23)
# vnum-keyed 40250 files whose rows an overlay (<stem>_zh_overlay.txt) replaces whole: items the server
# gives a new use (tools/server/hunt-set) get a new icon and description.
ROW_FILES = ("item_list.txt", "itemdesc.txt")
# itemdesc.txt columns (0-based: description, summary) translated by their English text (itemdesc_translate_zh.tsv).
ITEMDESC_COLUMNS = (2, 3)
# The zh client keeps 40250's Western tooltip path (uitooltip.SplitDescription): lines break at spaces, at most
# 35 bytes, and a token "|x" starts a new line. Chinese is wrapped here to the Eastern tooltip width instead
# (DESC_DEFAULT_MAX_COLS = 26 bytes, a CJK character is 2) and each line after the first starts with "|".
DESC_COLUMNS = 26
BREAK_AFTER = ",。、;:!?)”"
def keyed_lines(path: Path, encoding: str):
@@ -76,13 +83,88 @@ def convert_names(source: Path, destination: Path, overlay: Path = None):
return {"converted": converted, "overlaid": len(overlaid), "unrepresentable_vnums": unrepresentable}
def overlay_rows(path: Path, overlay: Path):
def wrap_lines(text: str, columns: int):
atoms = re.findall(r"[\x21-\x7e]+|\s+|.", text.strip())
lines, line, width, cut = [], [], 0, None
for atom in atoms:
size = len(atom.encode("gbk"))
if line and width + size > columns and atom not in BREAK_AFTER:
rest = []
if cut is not None and cut >= len(line) // 2:
rest, line = line[cut:], line[:cut]
lines.append("".join(line).strip())
while rest and rest[0].isspace():
rest.pop(0)
line, cut = rest, None
width = sum(len(a.encode("gbk")) for a in line)
if atom.isspace() and not line:
continue
line.append(atom)
width += size
if atom in BREAK_AFTER:
cut = len(line)
if line:
lines.append("".join(line).strip())
return lines
def wrap_description(text: str) -> str:
"""Chinese text as SplitDescription lines of at most DESC_COLUMNS bytes. ASCII runs (numbers, words) are not
split; a line ends after punctuation when one is in its second half, and punctuation never starts a line
(it may hang one character past the width, which the 35-byte Western limit still keeps on the line).
A last line of one or two characters is avoided by narrowing the lines when that needs no extra line."""
if all(ord(c) < 128 for c in text) or "|" in text:
return text
lines = wrap_lines(text, DESC_COLUMNS)
for columns in range(DESC_COLUMNS - 2, DESC_COLUMNS - 9, -2):
if len(lines) < 2 or len(lines[-1].encode("gbk")) > 6:
break
narrower = wrap_lines(text, columns)
if len(narrower) == len(lines):
lines = narrower
return " |".join(lines)
def translate_itemdesc(path: Path, table: Path):
"""itemdesc.txt with its description and summary columns replaced from the English→Chinese table (UTF-8,
English<TAB>Chinese) and wrapped; untranslated cells keep the English text."""
chinese = {}
for line in table.read_text(encoding="utf-8").splitlines():
if line and not line.startswith("#"):
english, text = line.split("\t")
chinese[english] = wrap_description(text)
rows, translated, english_rows = [], 0, []
for line in path.read_bytes().decode("cp1252").split("\n"):
parts = line.split("\t")
if parts[0].strip().isdigit():
left = False
for column in ITEMDESC_COLUMNS:
text = parts[column].strip() if column < len(parts) else ""
if not text:
continue
if text in chinese:
parts[column] = chinese[text] + ("\r" if parts[column].endswith("\r") else "")
translated += 1
else:
left = True
if left:
english_rows.append(parts[0].strip())
rows.append("\t".join(parts))
path.write_bytes("\n".join(rows).encode("gbk"))
return {"translated_cells": translated, "english_rows": english_rows}
def overlay_rows(path: Path, overlay: Path, wrap_columns=()):
"""Replace the rows of a vnum-keyed file by the overlay's (UTF-8, whole rows, no header); other rows keep
their bytes (the English files are cp1252)."""
their bytes (the English files are cp1252). wrap_columns (0-based) are wrapped like translated text."""
rows = {}
for line in overlay.read_text(encoding="utf-8").splitlines():
if line and not line.startswith("#"):
rows[line.split("\t", 1)[0]] = line.encode("gbk")
parts = line.split("\t")
for column in wrap_columns:
if column < len(parts):
parts[column] = wrap_description(parts[column])
rows[parts[0]] = "\t".join(parts).encode("gbk")
lines = path.read_bytes().split(b"\n")
replaced = []
for index, line in enumerate(lines):
@@ -236,10 +318,15 @@ def build(en: Path, zh: Path, output: Path, client_base: Path,
en / "skilldesc.txt", zh / "skilldesc.txt", locale_dir / "skilldesc.txt",
names_dir / "skilldesc_zh_overlay.txt" if names_dir else None)
table = names_dir / "itemdesc_translate_zh.tsv" if names_dir else None
if table and table.is_file():
report["files"]["itemdesc.txt translation"] = translate_itemdesc(locale_dir / "itemdesc.txt", table)
for filename in ROW_FILES:
overlay = names_dir / (Path(filename).stem + "_zh_overlay.txt") if names_dir else None
if overlay and overlay.is_file():
report["files"][filename] = overlay_rows(locale_dir / filename, overlay)
wrap = ITEMDESC_COLUMNS if filename == "itemdesc.txt" else ()
report["files"][filename] = overlay_rows(locale_dir / filename, overlay, wrap)
for folder in IMAGE_DIRS:
english_files = {path.name for path in (en / folder).iterdir()}