| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- """中文表达审核 (基准 = OEM 中文文档术语库 app_frontEnd/configs/terms/terms_baseline.yaml + 规则 app_frontEnd/configs/terms/jargon_rules.yaml).
- 机器给候选, 人裁 —— 不自动改文案。两类输出:
- ① 规则命中: 内部实现词/自造黑话/有更常用说法 (jargon_rules 逐条带建议写法与状态);
- ② 无根词: 界面文本里的领域词在四家 OEM 中文文档中**一次都没出现过** → 可能是自造词 (按出现次数排, 供人裁)。
- 用法: python scripts/audit_chinese_terms.py [--root .] [--out logs/terms_audit.json] [--md] [--strict]
- --strict: 规则状态=已裁改 的条目仍命中 → exit 2 (CI 闸); 默认只报不拦。"""
- from __future__ import annotations
- try:
- from app_common.app_common_guanlan.api import install_root as _install_root
- except ImportError: # 理论不可达;包结构异常时回退到按位置上跳
- from pathlib import Path as _P
- def _install_root(_f): return _P(_f).resolve().parents[4]
- import argparse, collections, json, re, sys, time
- from pathlib import Path
- import yaml
- CJK_RUN = re.compile(r"[一-鿿][一-鿿0-9A-Za-z·%°/\-]{1,60}")
- TERM = re.compile(r"[一-鿿]{2,8}")
- # ★P13(2026-10-01):审核目标随三栈重写更新 —— 前端语言包已由 Python `lang.py` 换成 Vue 语言包 `web/src/i18n/zh.ts`
- # (旧页与旧语言包按用户令退役;术语口径的审核点移到现行前端)。其余目标不变。
- DEFAULT_TARGETS = ["app_frontEnd/web/src/i18n/zh.ts",
- "app_backEnd/app_backEnd_guanlan/serve.py", "src/windcms/ui.py", "app_ontology/app_ontology_guanlan/ontology/fast_agent.py"]
- def strings_of(p: Path):
- """逐行取中文串 (含行号); 跳过注释行 —— 注释不进界面."""
- out = []
- for i, line in enumerate(p.read_text(encoding="utf-8", errors="replace").splitlines(), 1):
- s = line.strip()
- if s.startswith("#") or s.startswith("//") or s.startswith("*"): continue
- for m in CJK_RUN.finditer(line):
- t = m.group(0).strip()
- if len(t) >= 2: out.append((i, t))
- return out
- def main():
- ap = argparse.ArgumentParser(); ap.add_argument("--root", default="."); ap.add_argument("--baseline"); ap.add_argument("--rules")
- ap.add_argument("--targets", nargs="*"); ap.add_argument("--out", default="logs/audit/terms_audit.json") # 统一: 审计流水进 logs/audit/ (用户令 2)
- ap.add_argument("--md", action="store_true"); ap.add_argument("--strict", action="store_true"); ap.add_argument("--top", type=int, default=40); a = ap.parse_args()
- root = Path(a.root).resolve(); here = Path(__file__).resolve().parent
- def pick(arg, name):
- """基准/规则解析序: 显式参数 > 被审仓 app_frontEnd/configs/terms/ > 脚本同目录 (skill 自带正本)。
- app_frontEnd/configs/terms/ 那一项走 `src/paths.py` 的配置取用口 (P.config_dir('terms')),
- 不再手拼配置目录字符串 —— 用户令 2「统一配置」口径见 docs §11。
- """
- import sys as _sys
- _sys.path.insert(0, str(_install_root(__file__)))
- from src import paths as _P
- cands = ([Path(arg)] if arg else []) + [_P.config_dir('terms') / name, here / name, _install_root(__file__) / "configs" / "terms" / name] # config-path-allow: 末项是 skill 自带正本
- for c in cands:
- if c.exists(): return c
- raise SystemExit(f"找不到 {name} (给 --baseline/--rules 或放到 app_frontEnd/configs/terms/)")
- bp, rp = pick(a.baseline, "terms_baseline.yaml"), pick(a.rules, "jargon_rules.yaml")
- # 显示层映射: 会在渲染时被换掉的词, **用户看不到** → 只记为 fixed_by_display, 不算违规 (strict 不拦)。
- try:
- dm = yaml.safe_load(pick(None, "display_map.yaml").read_text(encoding="utf-8"))
- DISP = sorted([(str(i["from"]), str(i["to"])) for i in (dm.get("items") or [])], key=lambda kv: -len(kv[0]))
- except SystemExit: DISP = []
- def humanized(x):
- for u, v in DISP:
- if u in x: x = x.replace(u, v)
- return x
- base = yaml.safe_load(bp.read_text(encoding="utf-8")); known = {x["term"] for x in base["items"]}
- known_sub = set()
- for t in known: # 允许基准词作为子串命中 (如「主轴承温度」含「主轴承」「温度」)
- known_sub.add(t)
- _rdoc = yaml.safe_load(rp.read_text(encoding="utf-8")); rules = _rdoc["items"]
- # 豁免面: 模型提示词/内部注释/日志 —— 规则只约束**用户看得见的文本**
- EXEMPT_PATH = [e for e in (_rdoc.get("exempt") or []) if e.get("path")]
- EXEMPT_LINE = [(re.compile(e["line_re"]), e["why"]) for e in (_rdoc.get("exempt") or []) if e.get("line_re")]
- for r in rules: r["_re"] = re.compile(r["pattern"])
- targets = [root / t for t in (a.targets or DEFAULT_TARGETS)]
- hits, unknown = [], collections.Counter(); unknown_where = collections.defaultdict(list); n_strings = 0
- for p in targets:
- if not p.exists(): continue
- rel = p.relative_to(root).as_posix()
- ex_path = next((e["why"] for e in EXEMPT_PATH if e["path"] in rel), None)
- raw_lines = p.read_text(encoding="utf-8", errors="replace").splitlines()
- for ln, s in strings_of(p):
- n_strings += 1
- for r in rules:
- m = r["_re"].search(s)
- if not m: continue
- cur = raw_lines[ln - 1] if ln <= len(raw_lines) else ""
- ctx = "\n".join(raw_lines[max(0, ln - 3):ln + 2]) # 日志调用常跨行 (print(... 换行 file=sys.stderr))
- cpos = min([i for i in (cur.find("//"), cur.find("#")) if i >= 0], default=-1)
- in_comment = cpos >= 0 and cur.find(m.group(0)) > cpos # 行尾注释里的词不算界面文本
- ex = ex_path or ("行内注释" if in_comment else None) or next((w for rx, w in EXEMPT_LINE if rx.search(ctx)), None)
- hits.append(dict(file=rel, line=ln, text=s[:80], matched=m.group(0), exempt=ex,
- fixed_by_display=(humanized(s) != s), **{k: r[k] for k in ("class", "why", "suggest", "status", "pattern")}))
- for w in TERM.findall(s):
- if w in known_sub: continue
- if any(k in w for k in known_sub if len(k) >= 3): continue # 含已知术语的复合词不算无根
- unknown[w] += 1
- if len(unknown_where[w]) < 3: unknown_where[w].append(f"{p.name}:{ln}")
- ruled = [h for h in hits if h["status"] == "已裁改" and not h["fixed_by_display"] and not h["exempt"]] # 显示层能换掉的不算违规
- rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), root=str(root), baseline=str(bp), rules=str(rp), baseline_terms=len(known), n_strings=n_strings,
- n_rule_hits=len(hits), n_fixed_by_display=sum(1 for h in hits if h["fixed_by_display"]), n_exempt=sum(1 for h in hits if h["exempt"]), n_ruled_violations=len(ruled),
- by_class=dict(collections.Counter(h["class"] for h in hits)),
- rule_hits=hits, unrooted_terms=[dict(term=w, n=n, where=unknown_where[w]) for w, n in unknown.most_common(a.top)])
- o = root / a.out; o.parent.mkdir(parents=True, exist_ok=True); o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8")
- print(f"扫 {len(targets)} 文件 / {n_strings} 条中文串; 规则命中 {len(hits)} ({rep['by_class']}); 其中显示层已覆盖 {rep['n_fixed_by_display']}, 内部面豁免 {rep['n_exempt']}, 仍待处理 {sum(1 for h in hits if not h['fixed_by_display'] and not h['exempt'])}; 无根词 top{a.top} (共 {len(unknown)})")
- for h in [x for x in hits if not x["fixed_by_display"] and not x["exempt"]][:12]: print(f" [{h['class']}·{h['status']}] {h['file']}:{h['line']} 「{h['matched']}」 → 建议: {h['suggest'][:40]}")
- print(" 无根词:", [f"{w}×{n}" for w, n in unknown.most_common(15)])
- if a.md:
- L = [f"# 中文表达审核 · {rep['ts']}\n", f"基准 {len(known)} 词 (四家 OEM 中文文档实证) · 扫 {n_strings} 条界面中文串\n",
- "## 一 规则命中 (逐条裁: 改 / 豁免)\n", "| # | 类 | 文件:行 | 命中 | 现文 | 为什么难懂 | 建议写法 | 状态 |", "|---|---|---|---|---|---|---|---|"]
- for i, h in enumerate(hits, 1): L.append(f"| {i} | {h['class']} | {h['file']}:{h['line']} | {h['matched']} | {h['text'][:28]} | {h['why'][:34]} | {h['suggest'][:34]} | {h['status']} |")
- L += ["\n## 二 无根词 (OEM 中文文档零出现; 可能自造, 逐条裁)\n", "| # | 词 | 次数 | 出处 | 裁决 |", "|---|---|---|---|---|"]
- for i, u in enumerate(rep["unrooted_terms"], 1): L.append(f"| {i} | {u['term']} | {u['n']} | {' '.join(u['where'])} | |")
- m = o.with_suffix(".md"); m.write_text("\n".join(L) + "\n", encoding="utf-8"); print(" 屏:", m)
- if a.strict and ruled:
- print(f"✗ {len(ruled)} 条已裁定必改、且显示层也换不掉的表达仍在界面里:"); [print(" ", h["file"], h["line"], h["matched"]) for h in ruled[:10]]; return 2
- return 0
- if __name__ == "__main__":
- # 控制台可能是 GBK(中文 Windows 代码页 936): 正文里的 ✔ ✗ ✅ ⚠ 这类字符编不出来会抛
- # UnicodeEncodeError, 脚本干成了事却以退出码 1 结束(同类坑见 src/console.py)。降级为 '?' 而不是崩;
- # 不用 import 是为了兼顾 python -m 与直接当脚本跑两种启动方式。
- import sys as _sys
- for _s in (_sys.stdout, _sys.stderr):
- try: _s.reconfigure(errors='replace')
- except Exception: pass
- sys.exit(main())
|