audit_chinese_terms.py 8.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """中文表达审核 (基准 = OEM 中文文档术语库 configs/terms/terms_baseline.yaml + 规则 configs/terms/jargon_rules.yaml).
  4. 机器给候选, 人裁 —— 不自动改文案。两类输出:
  5. ① 规则命中: 内部实现词/自造黑话/有更常用说法 (jargon_rules 逐条带建议写法与状态);
  6. ② 无根词: 界面文本里的领域词在四家 OEM 中文文档中**一次都没出现过** → 可能是自造词 (按出现次数排, 供人裁)。
  7. 用法: python scripts/audit_chinese_terms.py [--root .] [--out logs/terms_audit.json] [--md] [--strict]
  8. --strict: 规则状态=已裁改 的条目仍命中 → exit 2 (CI 闸); 默认只报不拦。"""
  9. from __future__ import annotations
  10. import argparse, collections, json, re, sys, time
  11. from pathlib import Path
  12. import yaml
  13. CJK_RUN = re.compile(r"[一-鿿][一-鿿0-9A-Za-z·%°/\-]{1,60}")
  14. TERM = re.compile(r"[一-鿿]{2,8}")
  15. DEFAULT_TARGETS = ["src/windscada/lang.py", "src/windscada/ui/app.js", "scripts/windscada_serve.py", "src/windcms/ui.py", "src/ontology/fast_agent.py"]
  16. def strings_of(p: Path):
  17. """逐行取中文串 (含行号); 跳过注释行 —— 注释不进界面."""
  18. out = []
  19. for i, line in enumerate(p.read_text(encoding="utf-8", errors="replace").splitlines(), 1):
  20. s = line.strip()
  21. if s.startswith("#") or s.startswith("//") or s.startswith("*"): continue
  22. for m in CJK_RUN.finditer(line):
  23. t = m.group(0).strip()
  24. if len(t) >= 2: out.append((i, t))
  25. return out
  26. def main():
  27. ap = argparse.ArgumentParser(); ap.add_argument("--root", default="."); ap.add_argument("--baseline"); ap.add_argument("--rules")
  28. ap.add_argument("--targets", nargs="*"); ap.add_argument("--out", default="logs/terms_audit.json")
  29. ap.add_argument("--md", action="store_true"); ap.add_argument("--strict", action="store_true"); ap.add_argument("--top", type=int, default=40); a = ap.parse_args()
  30. root = Path(a.root).resolve(); here = Path(__file__).resolve().parent
  31. def pick(arg, name):
  32. """基准/规则解析序: 显式参数 > 被审仓 configs/terms/ > 脚本同目录 (skill 自带正本)."""
  33. for c in ([Path(arg)] if arg else []) + [root / "configs/terms" / name, here / name, here.parent / "configs/terms" / name]:
  34. if c.exists(): return c
  35. raise SystemExit(f"找不到 {name} (给 --baseline/--rules 或放到 configs/terms/)")
  36. bp, rp = pick(a.baseline, "terms_baseline.yaml"), pick(a.rules, "jargon_rules.yaml")
  37. # 显示层映射: 会在渲染时被换掉的词, **用户看不到** → 只记为 fixed_by_display, 不算违规 (strict 不拦)。
  38. try:
  39. dm = yaml.safe_load(pick(None, "display_map.yaml").read_text(encoding="utf-8"))
  40. DISP = sorted([(str(i["from"]), str(i["to"])) for i in (dm.get("items") or [])], key=lambda kv: -len(kv[0]))
  41. except SystemExit: DISP = []
  42. def humanized(x):
  43. for u, v in DISP:
  44. if u in x: x = x.replace(u, v)
  45. return x
  46. base = yaml.safe_load(bp.read_text(encoding="utf-8")); known = {x["term"] for x in base["items"]}
  47. known_sub = set()
  48. for t in known: # 允许基准词作为子串命中 (如「主轴承温度」含「主轴承」「温度」)
  49. known_sub.add(t)
  50. _rdoc = yaml.safe_load(rp.read_text(encoding="utf-8")); rules = _rdoc["items"]
  51. # 豁免面: 模型提示词/内部注释/日志 —— 规则只约束**用户看得见的文本**
  52. EXEMPT_PATH = [e for e in (_rdoc.get("exempt") or []) if e.get("path")]
  53. EXEMPT_LINE = [(re.compile(e["line_re"]), e["why"]) for e in (_rdoc.get("exempt") or []) if e.get("line_re")]
  54. for r in rules: r["_re"] = re.compile(r["pattern"])
  55. targets = [root / t for t in (a.targets or DEFAULT_TARGETS)]
  56. hits, unknown = [], collections.Counter(); unknown_where = collections.defaultdict(list); n_strings = 0
  57. for p in targets:
  58. if not p.exists(): continue
  59. rel = p.relative_to(root).as_posix()
  60. ex_path = next((e["why"] for e in EXEMPT_PATH if e["path"] in rel), None)
  61. raw_lines = p.read_text(encoding="utf-8", errors="replace").splitlines()
  62. for ln, s in strings_of(p):
  63. n_strings += 1
  64. for r in rules:
  65. m = r["_re"].search(s)
  66. if not m: continue
  67. cur = raw_lines[ln - 1] if ln <= len(raw_lines) else ""
  68. ctx = "\n".join(raw_lines[max(0, ln - 3):ln + 2]) # 日志调用常跨行 (print(... 换行 file=sys.stderr))
  69. cpos = min([i for i in (cur.find("//"), cur.find("#")) if i >= 0], default=-1)
  70. in_comment = cpos >= 0 and cur.find(m.group(0)) > cpos # 行尾注释里的词不算界面文本
  71. ex = ex_path or ("行内注释" if in_comment else None) or next((w for rx, w in EXEMPT_LINE if rx.search(ctx)), None)
  72. hits.append(dict(file=rel, line=ln, text=s[:80], matched=m.group(0), exempt=ex,
  73. fixed_by_display=(humanized(s) != s), **{k: r[k] for k in ("class", "why", "suggest", "status", "pattern")}))
  74. for w in TERM.findall(s):
  75. if w in known_sub: continue
  76. if any(k in w for k in known_sub if len(k) >= 3): continue # 含已知术语的复合词不算无根
  77. unknown[w] += 1
  78. if len(unknown_where[w]) < 3: unknown_where[w].append(f"{p.name}:{ln}")
  79. ruled = [h for h in hits if h["status"] == "已裁改" and not h["fixed_by_display"] and not h["exempt"]] # 显示层能换掉的不算违规
  80. rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), root=str(root), baseline=str(bp), rules=str(rp), baseline_terms=len(known), n_strings=n_strings,
  81. n_rule_hits=len(hits), n_fixed_by_display=sum(1 for h in hits if h["fixed_by_display"]), n_exempt=sum(1 for h in hits if h["exempt"]), n_ruled_violations=len(ruled),
  82. by_class=dict(collections.Counter(h["class"] for h in hits)),
  83. rule_hits=hits, unrooted_terms=[dict(term=w, n=n, where=unknown_where[w]) for w, n in unknown.most_common(a.top)])
  84. o = root / a.out; o.parent.mkdir(parents=True, exist_ok=True); o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8")
  85. print(f"扫 {len(targets)} 文件 / {n_strings} 条中文串; 规则命中 {len(hits)} ({rep['by_class']}); 其中显示层已覆盖 {rep['n_fixed_by_display']}, 内部面豁免 {rep['n_exempt']}, 仍待处理 {sum(1 for h in hits if not h['fixed_by_display'] and not h['exempt'])}; 无根词 top{a.top} (共 {len(unknown)})")
  86. for h in [x for x in hits if not x["fixed_by_display"] and not x["exempt"]][:12]: print(f" [{h['class']}·{h['status']}] {h['file']}:{h['line']} 「{h['matched']}」 → 建议: {h['suggest'][:40]}")
  87. print(" 无根词:", [f"{w}×{n}" for w, n in unknown.most_common(15)])
  88. if a.md:
  89. L = [f"# 中文表达审核 · {rep['ts']}\n", f"基准 {len(known)} 词 (四家 OEM 中文文档实证) · 扫 {n_strings} 条界面中文串\n",
  90. "## 一 规则命中 (逐条裁: 改 / 豁免)\n", "| # | 类 | 文件:行 | 命中 | 现文 | 为什么难懂 | 建议写法 | 状态 |", "|---|---|---|---|---|---|---|---|"]
  91. for i, h in enumerate(hits, 1): L.append(f"| {i} | {h['class']} | {h['file']}:{h['line']} | {h['matched']} | {h['text'][:28]} | {h['why'][:34]} | {h['suggest'][:34]} | {h['status']} |")
  92. L += ["\n## 二 无根词 (OEM 中文文档零出现; 可能自造, 逐条裁)\n", "| # | 词 | 次数 | 出处 | 裁决 |", "|---|---|---|---|---|"]
  93. for i, u in enumerate(rep["unrooted_terms"], 1): L.append(f"| {i} | {u['term']} | {u['n']} | {' '.join(u['where'])} | |")
  94. m = o.with_suffix(".md"); m.write_text("\n".join(L) + "\n", encoding="utf-8"); print(" 屏:", m)
  95. if a.strict and ruled:
  96. print(f"✗ {len(ruled)} 条已裁定必改、且显示层也换不掉的表达仍在界面里:"); [print(" ", h["file"], h["line"], h["matched"]) for h in ruled[:10]]; return 2
  97. return 0
  98. if __name__ == "__main__":
  99. sys.exit(main())