audit_chinese_terms.py 9.7 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """中文表达审核 (基准 = OEM 中文文档术语库 app_frontEnd/configs/terms/terms_baseline.yaml + 规则 app_frontEnd/configs/terms/jargon_rules.yaml).
  4. 机器给候选, 人裁 —— 不自动改文案。两类输出:
  5. ① 规则命中: 内部实现词/自造黑话/有更常用说法 (jargon_rules 逐条带建议写法与状态);
  6. ② 无根词: 界面文本里的领域词在四家 OEM 中文文档中**一次都没出现过** → 可能是自造词 (按出现次数排, 供人裁)。
  7. 用法: python scripts/audit_chinese_terms.py [--root .] [--out logs/terms_audit.json] [--md] [--strict]
  8. --strict: 规则状态=已裁改 的条目仍命中 → exit 2 (CI 闸); 默认只报不拦。"""
  9. from __future__ import annotations
  10. try:
  11. from app_common.app_common_guanlan.api import install_root as _install_root
  12. except ImportError: # 理论不可达;包结构异常时回退到按位置上跳
  13. from pathlib import Path as _P
  14. def _install_root(_f): return _P(_f).resolve().parents[4]
  15. import argparse, collections, json, re, sys, time
  16. from pathlib import Path
  17. import yaml
  18. CJK_RUN = re.compile(r"[一-鿿][一-鿿0-9A-Za-z·%°/\-]{1,60}")
  19. TERM = re.compile(r"[一-鿿]{2,8}")
  20. # ★P13(2026-10-01):审核目标随三栈重写更新 —— 前端语言包已由 Python `lang.py` 换成 Vue 语言包 `web/src/i18n/zh.ts`
  21. # (旧页与旧语言包按用户令退役;术语口径的审核点移到现行前端)。其余目标不变。
  22. DEFAULT_TARGETS = ["app_frontEnd/web/src/i18n/zh.ts",
  23. "app_backEnd/app_backEnd_guanlan/serve.py", "src/windcms/ui.py", "app_ontology/app_ontology_guanlan/ontology/fast_agent.py"]
  24. def strings_of(p: Path):
  25. """逐行取中文串 (含行号); 跳过注释行 —— 注释不进界面."""
  26. out = []
  27. for i, line in enumerate(p.read_text(encoding="utf-8", errors="replace").splitlines(), 1):
  28. s = line.strip()
  29. if s.startswith("#") or s.startswith("//") or s.startswith("*"): continue
  30. for m in CJK_RUN.finditer(line):
  31. t = m.group(0).strip()
  32. if len(t) >= 2: out.append((i, t))
  33. return out
  34. def main():
  35. ap = argparse.ArgumentParser(); ap.add_argument("--root", default="."); ap.add_argument("--baseline"); ap.add_argument("--rules")
  36. ap.add_argument("--targets", nargs="*"); ap.add_argument("--out", default="logs/audit/terms_audit.json") # 统一: 审计流水进 logs/audit/ (用户令 2)
  37. ap.add_argument("--md", action="store_true"); ap.add_argument("--strict", action="store_true"); ap.add_argument("--top", type=int, default=40); a = ap.parse_args()
  38. root = Path(a.root).resolve(); here = Path(__file__).resolve().parent
  39. def pick(arg, name):
  40. """基准/规则解析序: 显式参数 > 被审仓 app_frontEnd/configs/terms/ > 脚本同目录 (skill 自带正本)。
  41. app_frontEnd/configs/terms/ 那一项走 `src/paths.py` 的配置取用口 (P.config_dir('terms')),
  42. 不再手拼配置目录字符串 —— 用户令 2「统一配置」口径见 docs §11。
  43. """
  44. import sys as _sys
  45. _sys.path.insert(0, str(_install_root(__file__)))
  46. from src import paths as _P
  47. cands = ([Path(arg)] if arg else []) + [_P.config_dir('terms') / name, here / name, _install_root(__file__) / "configs" / "terms" / name] # config-path-allow: 末项是 skill 自带正本
  48. for c in cands:
  49. if c.exists(): return c
  50. raise SystemExit(f"找不到 {name} (给 --baseline/--rules 或放到 app_frontEnd/configs/terms/)")
  51. bp, rp = pick(a.baseline, "terms_baseline.yaml"), pick(a.rules, "jargon_rules.yaml")
  52. # 显示层映射: 会在渲染时被换掉的词, **用户看不到** → 只记为 fixed_by_display, 不算违规 (strict 不拦)。
  53. try:
  54. dm = yaml.safe_load(pick(None, "display_map.yaml").read_text(encoding="utf-8"))
  55. DISP = sorted([(str(i["from"]), str(i["to"])) for i in (dm.get("items") or [])], key=lambda kv: -len(kv[0]))
  56. except SystemExit: DISP = []
  57. def humanized(x):
  58. for u, v in DISP:
  59. if u in x: x = x.replace(u, v)
  60. return x
  61. base = yaml.safe_load(bp.read_text(encoding="utf-8")); known = {x["term"] for x in base["items"]}
  62. known_sub = set()
  63. for t in known: # 允许基准词作为子串命中 (如「主轴承温度」含「主轴承」「温度」)
  64. known_sub.add(t)
  65. _rdoc = yaml.safe_load(rp.read_text(encoding="utf-8")); rules = _rdoc["items"]
  66. # 豁免面: 模型提示词/内部注释/日志 —— 规则只约束**用户看得见的文本**
  67. EXEMPT_PATH = [e for e in (_rdoc.get("exempt") or []) if e.get("path")]
  68. EXEMPT_LINE = [(re.compile(e["line_re"]), e["why"]) for e in (_rdoc.get("exempt") or []) if e.get("line_re")]
  69. for r in rules: r["_re"] = re.compile(r["pattern"])
  70. targets = [root / t for t in (a.targets or DEFAULT_TARGETS)]
  71. hits, unknown = [], collections.Counter(); unknown_where = collections.defaultdict(list); n_strings = 0
  72. for p in targets:
  73. if not p.exists(): continue
  74. rel = p.relative_to(root).as_posix()
  75. ex_path = next((e["why"] for e in EXEMPT_PATH if e["path"] in rel), None)
  76. raw_lines = p.read_text(encoding="utf-8", errors="replace").splitlines()
  77. for ln, s in strings_of(p):
  78. n_strings += 1
  79. for r in rules:
  80. m = r["_re"].search(s)
  81. if not m: continue
  82. cur = raw_lines[ln - 1] if ln <= len(raw_lines) else ""
  83. ctx = "\n".join(raw_lines[max(0, ln - 3):ln + 2]) # 日志调用常跨行 (print(... 换行 file=sys.stderr))
  84. cpos = min([i for i in (cur.find("//"), cur.find("#")) if i >= 0], default=-1)
  85. in_comment = cpos >= 0 and cur.find(m.group(0)) > cpos # 行尾注释里的词不算界面文本
  86. ex = ex_path or ("行内注释" if in_comment else None) or next((w for rx, w in EXEMPT_LINE if rx.search(ctx)), None)
  87. hits.append(dict(file=rel, line=ln, text=s[:80], matched=m.group(0), exempt=ex,
  88. fixed_by_display=(humanized(s) != s), **{k: r[k] for k in ("class", "why", "suggest", "status", "pattern")}))
  89. for w in TERM.findall(s):
  90. if w in known_sub: continue
  91. if any(k in w for k in known_sub if len(k) >= 3): continue # 含已知术语的复合词不算无根
  92. unknown[w] += 1
  93. if len(unknown_where[w]) < 3: unknown_where[w].append(f"{p.name}:{ln}")
  94. ruled = [h for h in hits if h["status"] == "已裁改" and not h["fixed_by_display"] and not h["exempt"]] # 显示层能换掉的不算违规
  95. rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), root=str(root), baseline=str(bp), rules=str(rp), baseline_terms=len(known), n_strings=n_strings,
  96. n_rule_hits=len(hits), n_fixed_by_display=sum(1 for h in hits if h["fixed_by_display"]), n_exempt=sum(1 for h in hits if h["exempt"]), n_ruled_violations=len(ruled),
  97. by_class=dict(collections.Counter(h["class"] for h in hits)),
  98. rule_hits=hits, unrooted_terms=[dict(term=w, n=n, where=unknown_where[w]) for w, n in unknown.most_common(a.top)])
  99. o = root / a.out; o.parent.mkdir(parents=True, exist_ok=True); o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8")
  100. print(f"扫 {len(targets)} 文件 / {n_strings} 条中文串; 规则命中 {len(hits)} ({rep['by_class']}); 其中显示层已覆盖 {rep['n_fixed_by_display']}, 内部面豁免 {rep['n_exempt']}, 仍待处理 {sum(1 for h in hits if not h['fixed_by_display'] and not h['exempt'])}; 无根词 top{a.top} (共 {len(unknown)})")
  101. for h in [x for x in hits if not x["fixed_by_display"] and not x["exempt"]][:12]: print(f" [{h['class']}·{h['status']}] {h['file']}:{h['line']} 「{h['matched']}」 → 建议: {h['suggest'][:40]}")
  102. print(" 无根词:", [f"{w}×{n}" for w, n in unknown.most_common(15)])
  103. if a.md:
  104. L = [f"# 中文表达审核 · {rep['ts']}\n", f"基准 {len(known)} 词 (四家 OEM 中文文档实证) · 扫 {n_strings} 条界面中文串\n",
  105. "## 一 规则命中 (逐条裁: 改 / 豁免)\n", "| # | 类 | 文件:行 | 命中 | 现文 | 为什么难懂 | 建议写法 | 状态 |", "|---|---|---|---|---|---|---|---|"]
  106. for i, h in enumerate(hits, 1): L.append(f"| {i} | {h['class']} | {h['file']}:{h['line']} | {h['matched']} | {h['text'][:28]} | {h['why'][:34]} | {h['suggest'][:34]} | {h['status']} |")
  107. L += ["\n## 二 无根词 (OEM 中文文档零出现; 可能自造, 逐条裁)\n", "| # | 词 | 次数 | 出处 | 裁决 |", "|---|---|---|---|---|"]
  108. for i, u in enumerate(rep["unrooted_terms"], 1): L.append(f"| {i} | {u['term']} | {u['n']} | {' '.join(u['where'])} | |")
  109. m = o.with_suffix(".md"); m.write_text("\n".join(L) + "\n", encoding="utf-8"); print(" 屏:", m)
  110. if a.strict and ruled:
  111. print(f"✗ {len(ruled)} 条已裁定必改、且显示层也换不掉的表达仍在界面里:"); [print(" ", h["file"], h["line"], h["matched"]) for h in ruled[:10]]; return 2
  112. return 0
  113. if __name__ == "__main__":
  114. # 控制台可能是 GBK(中文 Windows 代码页 936): 正文里的 ✔ ✗ ✅ ⚠ 这类字符编不出来会抛
  115. # UnicodeEncodeError, 脚本干成了事却以退出码 1 结束(同类坑见 src/console.py)。降级为 '?' 而不是崩;
  116. # 不用 import 是为了兼顾 python -m 与直接当脚本跑两种启动方式。
  117. import sys as _sys
  118. for _s in (_sys.stdout, _sys.stderr):
  119. try: _s.reconfigure(errors='replace')
  120. except Exception: pass
  121. sys.exit(main())