import argparse import glob import json import os from collections import Counter def _unesc(s: str) -> str: if "\\" in s: return s.encode("utf-8").decode("unicode_escape") return s def _skip_ws_and_comments(text: str, i: int) -> int: n = len(text) while i < n: ch = text[i] if ch.isspace(): i += 1 continue if text.startswith("//", i): j = text.find("\n", i + 2) i = n if j == -1 else j + 1 continue if text.startswith("/*", i): j = text.find("*/", i + 2) i = n if j == -1 else j + 2 continue break return i def _parse_string_literal(text: str, i: int) -> tuple[str, int]: quote = text[i] i += 1 n = len(text) buf: list[str] = [] while i < n: ch = text[i] if ch == "\\" and i + 1 < n: buf.append(ch) buf.append(text[i + 1]) i += 2 continue if ch == quote: i += 1 break buf.append(ch) i += 1 raw = "".join(buf) return _unesc(raw), i def _parse_dart_map(file_path: str) -> dict[str, str]: with open(file_path, "r", encoding="utf-8") as f: text = f.read() start = text.find("{") end = text.rfind("}") if start == -1 or end == -1 or end <= start: return {} text = text[start + 1 : end] out: dict[str, str] = {} i = 0 n = len(text) while i < n: i = _skip_ws_and_comments(text, i) if i >= n: break if text[i] not in ("'", '"'): i += 1 continue key, i = _parse_string_literal(text, i) i = _skip_ws_and_comments(text, i) if i >= n or text[i] != ":": continue i += 1 i = _skip_ws_and_comments(text, i) if i >= n or text[i] not in ("'", '"'): continue parts: list[str] = [] while i < n and text[i] in ("'", '"'): part, i = _parse_string_literal(text, i) parts.append(part) i = _skip_ws_and_comments(text, i) out[key] = "".join(parts) return out def _ascii_letter_ratio(s: str) -> float: if not s: return 0.0 letters = sum(1 for ch in s if ("a" <= ch.lower() <= "z")) return letters / max(1, len(s)) def _looks_english_sentence(s: str) -> bool: if len(s.strip()) < 4: return False ratio = _ascii_letter_ratio(s) if ratio < 0.35: return False s_low = s.lower() common = [ "please", "failed", "success", "error", "connect", "connection", "device", "network", "download", "upload", "translate", "translation", "record", "permission", "enable", "disable", "check", "try again", "stats", "threshold", "cost", "estimated", "data volume", "source text", "characters", "duration", "session", "history", "cleanup", "export", "available", "package", "remaining", "quota", "feature description", "most-used", "audio", "file size", "bytes", "interrupt", "reward", "credits", "vip", "expiration", "membership", "subscription", "subscribe", "purchase", "buy", "upgrade", "yearly", "monthly", "template", "community", "coming soon", "comparison", "detailed", "information", "select image source", ] return any(w in s_low for w in common) def main() -> int: parser = argparse.ArgumentParser() parser.add_argument( "--dir", default=os.path.join("lib", "core", "translations", "language"), help="Directory containing language dart files", ) parser.add_argument( "--baseline", default="en_us.dart", help="Baseline dart file to compare values against", ) parser.add_argument( "--allowlist", default="", help="Path to JSON array of keys allowed to keep identical baseline values", ) parser.add_argument( "--mode", choices=["exact", "englishy"], default="englishy", help="exact: value equals baseline; englishy: value equals baseline and looks English", ) parser.add_argument( "--fail-on-missing", action="store_true", help="缺失 key 数超过 --allow-missing 时以非 0 退出(给 CI 用)", ) parser.add_argument( "--allow-missing", type=int, default=0, help="允许的缺失 key 总数基线;配合 --fail-on-missing 卡住「只增不减」", ) args = parser.parse_args() lang_dir = args.dir dart_files = sorted(glob.glob(os.path.join(lang_dir, "*.dart"))) if not dart_files: raise SystemExit(f"No dart files found in {lang_dir}") allowlist: set[str] = set() if args.allowlist: with open(args.allowlist, "r", encoding="utf-8") as f: allowlist = set(json.load(f)) parsed = {os.path.basename(p): _parse_dart_map(p) for p in dart_files} if args.baseline not in parsed: raise SystemExit(f"baseline not found: {args.baseline}") baseline = parsed[args.baseline] result: dict[str, dict] = {"baseline": args.baseline, "mode": args.mode, "files": {}} global_counter: Counter[str] = Counter() # 缺失 key 检查。 # # ⚠️ 这一段是 2026-09-05 补的,此前本工具**结构上无法发现缺失的 key**: # 下面那个循环遍历的是「该语言文件已有的 key」,且对 baseline 里没有的直接 # continue —— 一个 key 整个丢掉,它一声不吭。真实后果:tabTodo 在 30 个语言 # 文件里根本不存在,GetX 静默回退英文,谁也没发现,直到有人逐个文件去数。 # 以 zh_cn 为基准(它是文案的源语言,key 最全),不是以 en_us。 ref_name = "zh_cn.dart" if "zh_cn.dart" in parsed else args.baseline ref_keys = set(parsed[ref_name]) missing_counter: Counter[str] = Counter() for fn, kv in parsed.items(): if fn == args.baseline: continue issues: list[dict] = [] for k, v in kv.items(): if k in allowlist: continue bv = baseline.get(k) if bv is None: continue if v != bv: continue if args.mode == "englishy" and not _looks_english_sentence(v): continue issues.append({"key": k, "value": v}) global_counter[k] += 1 missing = sorted(ref_keys - set(kv)) for k in missing: missing_counter[k] += 1 result["files"][fn] = { "count": len(issues), "items": issues, "missing_count": len(missing), "missing": missing, } total_missing = sum(f["missing_count"] for f in result["files"].values()) result["summary"] = { "file_count": len(dart_files), "non_baseline_file_count": len(dart_files) - 1, "files_with_issues": sum(1 for f in result["files"].values() if f["count"] > 0), "total_issues": sum(f["count"] for f in result["files"].values()), "most_common_keys": [{"key": k, "count": c} for k, c in global_counter.most_common(50)], # 缺失统计 "missing_reference": ref_name, "files_with_missing": sum(1 for f in result["files"].values() if f["missing_count"] > 0), "total_missing": total_missing, "most_missing_keys": [{"key": k, "count": c} for k, c in missing_counter.most_common(50)], } print(json.dumps(result, ensure_ascii=False, indent=2)) # --fail-on-missing 给 CI 用:新增 key 只补了中文就提交时,这里非 0 退出。 # 不默认开启——仓库现存约 8000 处历史缺失,一上来就红会让人直接把检查关掉。 if args.fail_on_missing and total_missing > args.allow_missing: return 1 return 0 if __name__ == "__main__": raise SystemExit(main())