You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
220 lines
8.6 KiB
220 lines
8.6 KiB
#!/usr/bin/env python3
|
|
"""补齐语言文件里相对 en_us.dart 缺失的 key。
|
|
|
|
语言文件是 `const Map<String, String>`,key 同时存在 `'key'` 与 `"key"` 两种引号写法,
|
|
所以查重必须两种引号都认——const map 重复键要到 build 阶段才炸,analyze 不报。
|
|
|
|
用法(都在 apps/client 下执行):
|
|
|
|
python3 tools/i18n_fill_missing.py --report # 每个语言缺哪些 / 多哪些 key
|
|
python3 tools/i18n_fill_missing.py --dump-missing DIR # 把缺失 key 按语言写成 DIR/<lang>.json(key → 英文原文)
|
|
python3 tools/i18n_fill_missing.py --apply DIR # 读 DIR/<lang>.json(key → 译文)追加到各语言文件末尾
|
|
python3 tools/i18n_fill_missing.py --apply DIR --fallback-en # 译文文件里没有的 key 用英文原文补上
|
|
|
|
--apply 只追加**真正缺失**的 key(两种引号都查过),已有的一律不动;多余 key(enUS 没有的)只打印清单,不删。
|
|
译文里的占位符(@name / %s / {0} / \\n)会与英文原文比对,不一致的 key 会拒绝写入并列出来。
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import glob
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
LANG_DIR = os.path.join("lib", "core", "translations", "language")
|
|
BASELINE = "en_us.dart"
|
|
|
|
# 一条 entry:key 字面量 + 冒号 + 一个或多个相邻字符串字面量(Dart 相邻字符串会拼接)
|
|
_STR = r"""(?:'(?:\\.|[^'\\])*'|"(?:\\.|[^"\\])*")"""
|
|
_ENTRY_RE = re.compile(
|
|
r"^[ \t]*(?P<key>" + _STR + r")\s*:\s*(?P<val>(?:" + _STR + r"\s*)+),",
|
|
re.M,
|
|
)
|
|
_KEY_RE = re.compile(r"^[ \t]*(['\"])((?:\\.|(?!\1).)*)\1\s*:", re.M)
|
|
_PLACEHOLDER_RE = re.compile(r"@\w+|%[sd]|\{\w+\}|\n")
|
|
|
|
|
|
def _unquote(lit: str) -> str:
|
|
"""把一个 Dart 字符串字面量还原成值(处理 \\' \\" \\\\ \\n \\t \\$ \\uXXXX)。"""
|
|
body = lit[1:-1]
|
|
out: list[str] = []
|
|
i = 0
|
|
while i < len(body):
|
|
ch = body[i]
|
|
if ch == "\\" and i + 1 < len(body):
|
|
nxt = body[i + 1]
|
|
if nxt == "u":
|
|
# \uXXXX 或 \u{X...}
|
|
if i + 2 < len(body) and body[i + 2] == "{":
|
|
j = body.index("}", i + 3)
|
|
out.append(chr(int(body[i + 3 : j], 16)))
|
|
i = j + 1
|
|
else:
|
|
out.append(chr(int(body[i + 2 : i + 6], 16)))
|
|
i += 6
|
|
continue
|
|
out.append({"n": "\n", "t": "\t"}.get(nxt, nxt))
|
|
i += 2
|
|
continue
|
|
out.append(ch)
|
|
i += 1
|
|
return "".join(out)
|
|
|
|
|
|
def _quote(value: str) -> str:
|
|
"""把值写成单引号 Dart 字面量;`$` 要转义,否则被当成插值。"""
|
|
s = (
|
|
value.replace("\\", "\\\\")
|
|
.replace("'", "\\'")
|
|
.replace("$", "\\$")
|
|
.replace("\n", "\\n")
|
|
.replace("\t", "\\t")
|
|
)
|
|
return f"'{s}'"
|
|
|
|
|
|
def parse_file(path: str) -> tuple[dict[str, str], list[str]]:
|
|
"""返回 (key → 值, 重复出现的 key 列表)。两种引号的 key 一起算。"""
|
|
with open(path, encoding="utf-8") as f:
|
|
text = f.read()
|
|
values: dict[str, str] = {}
|
|
seen: dict[str, int] = {}
|
|
for m in _ENTRY_RE.finditer(text):
|
|
key = _unquote(m.group("key"))
|
|
parts = re.findall(_STR, m.group("val"))
|
|
values[key] = "".join(_unquote(p) for p in parts)
|
|
seen[key] = seen.get(key, 0) + 1
|
|
# _ENTRY_RE 只认值是字符串字面量的行;再用纯 key 正则兜一遍,防止漏掉写法怪异的行
|
|
for m in _KEY_RE.finditer(text):
|
|
key = _unquote(m.group(1) + m.group(2) + m.group(1))
|
|
if key not in seen:
|
|
seen[key] = 1
|
|
values.setdefault(key, "")
|
|
dups = sorted(k for k, n in seen.items() if n > 1)
|
|
return values, dups
|
|
|
|
|
|
def placeholders(s: str) -> list[str]:
|
|
return sorted(_PLACEHOLDER_RE.findall(s))
|
|
|
|
|
|
def lang_files() -> list[str]:
|
|
return sorted(
|
|
p for p in glob.glob(os.path.join(LANG_DIR, "*.dart")) if os.path.basename(p) != BASELINE
|
|
)
|
|
|
|
|
|
def lang_name(path: str) -> str:
|
|
return os.path.splitext(os.path.basename(path))[0]
|
|
|
|
|
|
def append_entries(path: str, entries: list[tuple[str, str]]) -> None:
|
|
with open(path, encoding="utf-8") as f:
|
|
text = f.read()
|
|
end = text.rfind("};")
|
|
if end == -1:
|
|
raise SystemExit(f"{path}: 找不到 map 结尾 '}};'")
|
|
block = "".join(f" {_quote(k)}: {_quote(v)},\n" for k, v in entries)
|
|
before = text[:end]
|
|
if not before.endswith("\n"):
|
|
before += "\n"
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
f.write(before + block + text[end:])
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
ap.add_argument("--report", action="store_true", help="只输出每个语言缺失/多余 key 的统计")
|
|
ap.add_argument("--dump-missing", metavar="DIR", help="把缺失 key(key → 英文原文)按语言写到 DIR/<lang>.json")
|
|
ap.add_argument("--apply", metavar="DIR", help="读 DIR/<lang>.json(key → 译文)追加到语言文件")
|
|
ap.add_argument("--fallback-en", action="store_true", help="--apply 时译文缺的 key 用英文原文补上")
|
|
ap.add_argument("--only", nargs="*", help="只处理这些语言(如 es_es fr_fr)")
|
|
args = ap.parse_args()
|
|
|
|
base_path = os.path.join(LANG_DIR, BASELINE)
|
|
base, base_dups = parse_file(base_path)
|
|
if base_dups:
|
|
print(f"[警告] {BASELINE} 自身有重复 key: {base_dups}")
|
|
print(f"{BASELINE}: {len(base)} keys")
|
|
|
|
files = lang_files()
|
|
if args.only:
|
|
files = [p for p in files if lang_name(p) in set(args.only)]
|
|
|
|
total_added = 0
|
|
english_kept: dict[str, list[str]] = {}
|
|
rejected: dict[str, list[str]] = {}
|
|
extras_all: dict[str, list[str]] = {}
|
|
|
|
for path in files:
|
|
lang = lang_name(path)
|
|
cur, dups = parse_file(path)
|
|
if dups:
|
|
print(f"[警告] {lang}: 文件内已有重复 key: {dups}")
|
|
missing = [k for k in base if k not in cur]
|
|
extra = [k for k in cur if k not in base]
|
|
if extra:
|
|
extras_all[lang] = extra
|
|
|
|
if args.report or (not args.dump_missing and not args.apply):
|
|
print(f"{lang}: {len(cur)} keys, 缺 {len(missing)}, 多 {len(extra)}")
|
|
continue
|
|
|
|
if args.dump_missing:
|
|
os.makedirs(args.dump_missing, exist_ok=True)
|
|
with open(os.path.join(args.dump_missing, f"{lang}.json"), "w", encoding="utf-8") as f:
|
|
json.dump({k: base[k] for k in missing}, f, ensure_ascii=False, indent=2)
|
|
print(f"{lang}: 写出 {len(missing)} 个缺失 key")
|
|
continue
|
|
|
|
if args.apply:
|
|
tpath = os.path.join(args.apply, f"{lang}.json")
|
|
trans: dict[str, str] = {}
|
|
if os.path.exists(tpath):
|
|
with open(tpath, encoding="utf-8") as f:
|
|
trans = json.load(f)
|
|
entries: list[tuple[str, str]] = []
|
|
for k in missing:
|
|
v = trans.get(k)
|
|
if v is None or not str(v).strip():
|
|
if not args.fallback_en:
|
|
rejected.setdefault(lang, []).append(f"{k} (无译文)")
|
|
continue
|
|
v = base[k]
|
|
v = str(v)
|
|
if placeholders(v) != placeholders(base[k]):
|
|
rejected.setdefault(lang, []).append(
|
|
f"{k} (占位符不一致: {placeholders(v)} vs {placeholders(base[k])})"
|
|
)
|
|
continue
|
|
if v == base[k] and re.search(r"[A-Za-z]{3,}", v):
|
|
english_kept.setdefault(lang, []).append(k)
|
|
entries.append((k, v))
|
|
if entries:
|
|
append_entries(path, entries)
|
|
total_added += len(entries)
|
|
print(f"{lang}: 追加 {len(entries)} 个 key" + (f",跳过 {len(rejected.get(lang, []))}" if lang in rejected else ""))
|
|
|
|
if extras_all:
|
|
print("\n== enUS 没有、语言文件多出来的 key(未删除,仅列出)==")
|
|
for lang, ks in extras_all.items():
|
|
print(f"{lang} ({len(ks)}): {', '.join(ks)}")
|
|
if rejected:
|
|
print("\n== 未写入的 key ==")
|
|
for lang, ks in rejected.items():
|
|
for k in ks:
|
|
print(f"{lang}: {k}")
|
|
if english_kept:
|
|
print("\n== 与英文原文相同(保留英文)的 key ==")
|
|
for lang, ks in english_kept.items():
|
|
print(f"{lang} ({len(ks)}): {', '.join(ks)}")
|
|
if args.apply:
|
|
print(f"\n共追加 {total_added} 条")
|
|
return 1 if rejected else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|
|
|