#!/usr/bin/env python3 # -*- coding: utf-8 -*- """中文表达审核 (基准 = OEM 中文文档术语库 configs/terms/terms_baseline.yaml + 规则 configs/terms/jargon_rules.yaml). 机器给候选, 人裁 —— 不自动改文案。两类输出: ① 规则命中: 内部实现词/自造黑话/有更常用说法 (jargon_rules 逐条带建议写法与状态); ② 无根词: 界面文本里的领域词在四家 OEM 中文文档中**一次都没出现过** → 可能是自造词 (按出现次数排, 供人裁)。 用法: python scripts/audit_chinese_terms.py [--root .] [--out logs/terms_audit.json] [--md] [--strict] --strict: 规则状态=已裁改 的条目仍命中 → exit 2 (CI 闸); 默认只报不拦。""" from __future__ import annotations import argparse, collections, json, re, sys, time from pathlib import Path import yaml CJK_RUN = re.compile(r"[一-鿿][一-鿿0-9A-Za-z·%°/\-]{1,60}") TERM = re.compile(r"[一-鿿]{2,8}") DEFAULT_TARGETS = ["src/windscada/lang.py", "src/windscada/ui/app.js", "scripts/windscada_serve.py", "src/windcms/ui.py", "src/ontology/fast_agent.py"] def strings_of(p: Path): """逐行取中文串 (含行号); 跳过注释行 —— 注释不进界面.""" out = [] for i, line in enumerate(p.read_text(encoding="utf-8", errors="replace").splitlines(), 1): s = line.strip() if s.startswith("#") or s.startswith("//") or s.startswith("*"): continue for m in CJK_RUN.finditer(line): t = m.group(0).strip() if len(t) >= 2: out.append((i, t)) return out def main(): ap = argparse.ArgumentParser(); ap.add_argument("--root", default="."); ap.add_argument("--baseline"); ap.add_argument("--rules") ap.add_argument("--targets", nargs="*"); ap.add_argument("--out", default="logs/terms_audit.json") ap.add_argument("--md", action="store_true"); ap.add_argument("--strict", action="store_true"); ap.add_argument("--top", type=int, default=40); a = ap.parse_args() root = Path(a.root).resolve(); here = Path(__file__).resolve().parent def pick(arg, name): """基准/规则解析序: 显式参数 > 被审仓 configs/terms/ > 脚本同目录 (skill 自带正本).""" for c in ([Path(arg)] if arg else []) + [root / "configs/terms" / name, here / name, here.parent / "configs/terms" / name]: if c.exists(): return c raise SystemExit(f"找不到 {name} (给 --baseline/--rules 或放到 configs/terms/)") bp, rp = pick(a.baseline, "terms_baseline.yaml"), pick(a.rules, "jargon_rules.yaml") # 显示层映射: 会在渲染时被换掉的词, **用户看不到** → 只记为 fixed_by_display, 不算违规 (strict 不拦)。 try: dm = yaml.safe_load(pick(None, "display_map.yaml").read_text(encoding="utf-8")) DISP = sorted([(str(i["from"]), str(i["to"])) for i in (dm.get("items") or [])], key=lambda kv: -len(kv[0])) except SystemExit: DISP = [] def humanized(x): for u, v in DISP: if u in x: x = x.replace(u, v) return x base = yaml.safe_load(bp.read_text(encoding="utf-8")); known = {x["term"] for x in base["items"]} known_sub = set() for t in known: # 允许基准词作为子串命中 (如「主轴承温度」含「主轴承」「温度」) known_sub.add(t) _rdoc = yaml.safe_load(rp.read_text(encoding="utf-8")); rules = _rdoc["items"] # 豁免面: 模型提示词/内部注释/日志 —— 规则只约束**用户看得见的文本** EXEMPT_PATH = [e for e in (_rdoc.get("exempt") or []) if e.get("path")] EXEMPT_LINE = [(re.compile(e["line_re"]), e["why"]) for e in (_rdoc.get("exempt") or []) if e.get("line_re")] for r in rules: r["_re"] = re.compile(r["pattern"]) targets = [root / t for t in (a.targets or DEFAULT_TARGETS)] hits, unknown = [], collections.Counter(); unknown_where = collections.defaultdict(list); n_strings = 0 for p in targets: if not p.exists(): continue rel = p.relative_to(root).as_posix() ex_path = next((e["why"] for e in EXEMPT_PATH if e["path"] in rel), None) raw_lines = p.read_text(encoding="utf-8", errors="replace").splitlines() for ln, s in strings_of(p): n_strings += 1 for r in rules: m = r["_re"].search(s) if not m: continue cur = raw_lines[ln - 1] if ln <= len(raw_lines) else "" ctx = "\n".join(raw_lines[max(0, ln - 3):ln + 2]) # 日志调用常跨行 (print(... 换行 file=sys.stderr)) cpos = min([i for i in (cur.find("//"), cur.find("#")) if i >= 0], default=-1) in_comment = cpos >= 0 and cur.find(m.group(0)) > cpos # 行尾注释里的词不算界面文本 ex = ex_path or ("行内注释" if in_comment else None) or next((w for rx, w in EXEMPT_LINE if rx.search(ctx)), None) hits.append(dict(file=rel, line=ln, text=s[:80], matched=m.group(0), exempt=ex, fixed_by_display=(humanized(s) != s), **{k: r[k] for k in ("class", "why", "suggest", "status", "pattern")})) for w in TERM.findall(s): if w in known_sub: continue if any(k in w for k in known_sub if len(k) >= 3): continue # 含已知术语的复合词不算无根 unknown[w] += 1 if len(unknown_where[w]) < 3: unknown_where[w].append(f"{p.name}:{ln}") ruled = [h for h in hits if h["status"] == "已裁改" and not h["fixed_by_display"] and not h["exempt"]] # 显示层能换掉的不算违规 rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), root=str(root), baseline=str(bp), rules=str(rp), baseline_terms=len(known), n_strings=n_strings, n_rule_hits=len(hits), n_fixed_by_display=sum(1 for h in hits if h["fixed_by_display"]), n_exempt=sum(1 for h in hits if h["exempt"]), n_ruled_violations=len(ruled), by_class=dict(collections.Counter(h["class"] for h in hits)), rule_hits=hits, unrooted_terms=[dict(term=w, n=n, where=unknown_where[w]) for w, n in unknown.most_common(a.top)]) o = root / a.out; o.parent.mkdir(parents=True, exist_ok=True); o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8") print(f"扫 {len(targets)} 文件 / {n_strings} 条中文串; 规则命中 {len(hits)} ({rep['by_class']}); 其中显示层已覆盖 {rep['n_fixed_by_display']}, 内部面豁免 {rep['n_exempt']}, 仍待处理 {sum(1 for h in hits if not h['fixed_by_display'] and not h['exempt'])}; 无根词 top{a.top} (共 {len(unknown)})") for h in [x for x in hits if not x["fixed_by_display"] and not x["exempt"]][:12]: print(f" [{h['class']}·{h['status']}] {h['file']}:{h['line']} 「{h['matched']}」 → 建议: {h['suggest'][:40]}") print(" 无根词:", [f"{w}×{n}" for w, n in unknown.most_common(15)]) if a.md: L = [f"# 中文表达审核 · {rep['ts']}\n", f"基准 {len(known)} 词 (四家 OEM 中文文档实证) · 扫 {n_strings} 条界面中文串\n", "## 一 规则命中 (逐条裁: 改 / 豁免)\n", "| # | 类 | 文件:行 | 命中 | 现文 | 为什么难懂 | 建议写法 | 状态 |", "|---|---|---|---|---|---|---|---|"] for i, h in enumerate(hits, 1): L.append(f"| {i} | {h['class']} | {h['file']}:{h['line']} | {h['matched']} | {h['text'][:28]} | {h['why'][:34]} | {h['suggest'][:34]} | {h['status']} |") L += ["\n## 二 无根词 (OEM 中文文档零出现; 可能自造, 逐条裁)\n", "| # | 词 | 次数 | 出处 | 裁决 |", "|---|---|---|---|---|"] for i, u in enumerate(rep["unrooted_terms"], 1): L.append(f"| {i} | {u['term']} | {u['n']} | {' '.join(u['where'])} | |") m = o.with_suffix(".md"); m.write_text("\n".join(L) + "\n", encoding="utf-8"); print(" 屏:", m) if a.strict and ruled: print(f"✗ {len(ruled)} 条已裁定必改、且显示层也换不掉的表达仍在界面里:"); [print(" ", h["file"], h["line"], h["matched"]) for h in ruled[:10]]; return 2 return 0 if __name__ == "__main__": # 控制台可能是 GBK(中文 Windows 代码页 936): 正文里的 ✔ ✗ ✅ ⚠ 这类字符编不出来会抛 # UnicodeEncodeError, 脚本干成了事却以退出码 1 结束(同类坑见 src/console.py)。降级为 '?' 而不是崩; # 不用 import 是为了兼顾 python -m 与直接当脚本跑两种启动方式。 import sys as _sys for _s in (_sys.stdout, _sys.stderr): try: _s.reconfigure(errors='replace') except Exception: pass sys.exit(main())