audit_chinese_terms.py 9.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """中文表达审核 (基准 = OEM 中文文档术语库 configs/terms/terms_baseline.yaml + 规则 configs/terms/jargon_rules.yaml).
  4. 机器给候选, 人裁 —— 不自动改文案。两类输出:
  5. ① 规则命中: 内部实现词/自造黑话/有更常用说法 (jargon_rules 逐条带建议写法与状态);
  6. ② 无根词: 界面文本里的领域词在四家 OEM 中文文档中**一次都没出现过** → 可能是自造词 (按出现次数排, 供人裁)。
  7. 用法: python scripts/audit_chinese_terms.py [--root .] [--out logs/terms_audit.json] [--md] [--strict]
  8. --strict: 规则状态=已裁改 的条目仍命中 → exit 2 (CI 闸); 默认只报不拦。"""
  9. from __future__ import annotations
  10. import argparse, collections, json, re, sys, time
  11. from pathlib import Path
  12. import yaml
  13. CJK_RUN = re.compile(r"[一-鿿][一-鿿0-9A-Za-z·%°/\-]{1,60}")
  14. TERM = re.compile(r"[一-鿿]{2,8}")
  15. DEFAULT_TARGETS = ["src/windscada/lang.py", "src/windscada/ui/app.js", "scripts/windscada_serve.py", "src/windcms/ui.py", "src/ontology/fast_agent.py"]
  16. def strings_of(p: Path):
  17. """逐行取中文串 (含行号); 跳过注释行 —— 注释不进界面."""
  18. out = []
  19. for i, line in enumerate(p.read_text(encoding="utf-8", errors="replace").splitlines(), 1):
  20. s = line.strip()
  21. if s.startswith("#") or s.startswith("//") or s.startswith("*"): continue
  22. for m in CJK_RUN.finditer(line):
  23. t = m.group(0).strip()
  24. if len(t) >= 2: out.append((i, t))
  25. return out
  26. def main():
  27. ap = argparse.ArgumentParser(); ap.add_argument("--root", default="."); ap.add_argument("--baseline"); ap.add_argument("--rules")
  28. ap.add_argument("--targets", nargs="*"); ap.add_argument("--out", default="logs/audit/terms_audit.json") # 统一: 审计流水进 logs/audit/ (用户令 2)
  29. ap.add_argument("--md", action="store_true"); ap.add_argument("--strict", action="store_true"); ap.add_argument("--top", type=int, default=40); a = ap.parse_args()
  30. root = Path(a.root).resolve(); here = Path(__file__).resolve().parent
  31. def pick(arg, name):
  32. """基准/规则解析序: 显式参数 > 被审仓 configs/terms/ > 脚本同目录 (skill 自带正本)。
  33. configs/terms/ 那一项走 `src/paths.py` 的配置取用口 (P.config_dir('terms')),
  34. 不再手拼配置目录字符串 —— 用户令 2「统一配置」口径见 docs §11。
  35. """
  36. import sys as _sys
  37. _sys.path.insert(0, str(here.parent))
  38. from src import paths as _P
  39. cands = ([Path(arg)] if arg else []) + [_P.config_dir('terms') / name, here / name, here.parent / "configs" / "terms" / name] # config-path-allow: 末项是 skill 自带正本
  40. for c in cands:
  41. if c.exists(): return c
  42. raise SystemExit(f"找不到 {name} (给 --baseline/--rules 或放到 configs/terms/)")
  43. bp, rp = pick(a.baseline, "terms_baseline.yaml"), pick(a.rules, "jargon_rules.yaml")
  44. # 显示层映射: 会在渲染时被换掉的词, **用户看不到** → 只记为 fixed_by_display, 不算违规 (strict 不拦)。
  45. try:
  46. dm = yaml.safe_load(pick(None, "display_map.yaml").read_text(encoding="utf-8"))
  47. DISP = sorted([(str(i["from"]), str(i["to"])) for i in (dm.get("items") or [])], key=lambda kv: -len(kv[0]))
  48. except SystemExit: DISP = []
  49. def humanized(x):
  50. for u, v in DISP:
  51. if u in x: x = x.replace(u, v)
  52. return x
  53. base = yaml.safe_load(bp.read_text(encoding="utf-8")); known = {x["term"] for x in base["items"]}
  54. known_sub = set()
  55. for t in known: # 允许基准词作为子串命中 (如「主轴承温度」含「主轴承」「温度」)
  56. known_sub.add(t)
  57. _rdoc = yaml.safe_load(rp.read_text(encoding="utf-8")); rules = _rdoc["items"]
  58. # 豁免面: 模型提示词/内部注释/日志 —— 规则只约束**用户看得见的文本**
  59. EXEMPT_PATH = [e for e in (_rdoc.get("exempt") or []) if e.get("path")]
  60. EXEMPT_LINE = [(re.compile(e["line_re"]), e["why"]) for e in (_rdoc.get("exempt") or []) if e.get("line_re")]
  61. for r in rules: r["_re"] = re.compile(r["pattern"])
  62. targets = [root / t for t in (a.targets or DEFAULT_TARGETS)]
  63. hits, unknown = [], collections.Counter(); unknown_where = collections.defaultdict(list); n_strings = 0
  64. for p in targets:
  65. if not p.exists(): continue
  66. rel = p.relative_to(root).as_posix()
  67. ex_path = next((e["why"] for e in EXEMPT_PATH if e["path"] in rel), None)
  68. raw_lines = p.read_text(encoding="utf-8", errors="replace").splitlines()
  69. for ln, s in strings_of(p):
  70. n_strings += 1
  71. for r in rules:
  72. m = r["_re"].search(s)
  73. if not m: continue
  74. cur = raw_lines[ln - 1] if ln <= len(raw_lines) else ""
  75. ctx = "\n".join(raw_lines[max(0, ln - 3):ln + 2]) # 日志调用常跨行 (print(... 换行 file=sys.stderr))
  76. cpos = min([i for i in (cur.find("//"), cur.find("#")) if i >= 0], default=-1)
  77. in_comment = cpos >= 0 and cur.find(m.group(0)) > cpos # 行尾注释里的词不算界面文本
  78. ex = ex_path or ("行内注释" if in_comment else None) or next((w for rx, w in EXEMPT_LINE if rx.search(ctx)), None)
  79. hits.append(dict(file=rel, line=ln, text=s[:80], matched=m.group(0), exempt=ex,
  80. fixed_by_display=(humanized(s) != s), **{k: r[k] for k in ("class", "why", "suggest", "status", "pattern")}))
  81. for w in TERM.findall(s):
  82. if w in known_sub: continue
  83. if any(k in w for k in known_sub if len(k) >= 3): continue # 含已知术语的复合词不算无根
  84. unknown[w] += 1
  85. if len(unknown_where[w]) < 3: unknown_where[w].append(f"{p.name}:{ln}")
  86. ruled = [h for h in hits if h["status"] == "已裁改" and not h["fixed_by_display"] and not h["exempt"]] # 显示层能换掉的不算违规
  87. rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), root=str(root), baseline=str(bp), rules=str(rp), baseline_terms=len(known), n_strings=n_strings,
  88. n_rule_hits=len(hits), n_fixed_by_display=sum(1 for h in hits if h["fixed_by_display"]), n_exempt=sum(1 for h in hits if h["exempt"]), n_ruled_violations=len(ruled),
  89. by_class=dict(collections.Counter(h["class"] for h in hits)),
  90. rule_hits=hits, unrooted_terms=[dict(term=w, n=n, where=unknown_where[w]) for w, n in unknown.most_common(a.top)])
  91. o = root / a.out; o.parent.mkdir(parents=True, exist_ok=True); o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8")
  92. print(f"扫 {len(targets)} 文件 / {n_strings} 条中文串; 规则命中 {len(hits)} ({rep['by_class']}); 其中显示层已覆盖 {rep['n_fixed_by_display']}, 内部面豁免 {rep['n_exempt']}, 仍待处理 {sum(1 for h in hits if not h['fixed_by_display'] and not h['exempt'])}; 无根词 top{a.top} (共 {len(unknown)})")
  93. for h in [x for x in hits if not x["fixed_by_display"] and not x["exempt"]][:12]: print(f" [{h['class']}·{h['status']}] {h['file']}:{h['line']} 「{h['matched']}」 → 建议: {h['suggest'][:40]}")
  94. print(" 无根词:", [f"{w}×{n}" for w, n in unknown.most_common(15)])
  95. if a.md:
  96. L = [f"# 中文表达审核 · {rep['ts']}\n", f"基准 {len(known)} 词 (四家 OEM 中文文档实证) · 扫 {n_strings} 条界面中文串\n",
  97. "## 一 规则命中 (逐条裁: 改 / 豁免)\n", "| # | 类 | 文件:行 | 命中 | 现文 | 为什么难懂 | 建议写法 | 状态 |", "|---|---|---|---|---|---|---|---|"]
  98. for i, h in enumerate(hits, 1): L.append(f"| {i} | {h['class']} | {h['file']}:{h['line']} | {h['matched']} | {h['text'][:28]} | {h['why'][:34]} | {h['suggest'][:34]} | {h['status']} |")
  99. L += ["\n## 二 无根词 (OEM 中文文档零出现; 可能自造, 逐条裁)\n", "| # | 词 | 次数 | 出处 | 裁决 |", "|---|---|---|---|---|"]
  100. for i, u in enumerate(rep["unrooted_terms"], 1): L.append(f"| {i} | {u['term']} | {u['n']} | {' '.join(u['where'])} | |")
  101. m = o.with_suffix(".md"); m.write_text("\n".join(L) + "\n", encoding="utf-8"); print(" 屏:", m)
  102. if a.strict and ruled:
  103. print(f"✗ {len(ruled)} 条已裁定必改、且显示层也换不掉的表达仍在界面里:"); [print(" ", h["file"], h["line"], h["matched"]) for h in ruled[:10]]; return 2
  104. return 0
  105. if __name__ == "__main__":
  106. # 控制台可能是 GBK(中文 Windows 代码页 936): 正文里的 ✔ ✗ ✅ ⚠ 这类字符编不出来会抛
  107. # UnicodeEncodeError, 脚本干成了事却以退出码 1 结束(同类坑见 src/console.py)。降级为 '?' 而不是崩;
  108. # 不用 import 是为了兼顾 python -m 与直接当脚本跑两种启动方式。
  109. import sys as _sys
  110. for _s in (_sys.stdout, _sys.stderr):
  111. try: _s.reconfigure(errors='replace')
  112. except Exception: pass
  113. sys.exit(main())