build_oem_lexicon.py 5.4 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """OEM 中文术语库构建 (用户令 2026-09-08: "找 GE、西门子、维斯塔斯、歌美飒的中文文档, 建立词库, 审核中文表达").
  4. 只抽**术语与词频**, 不复制原文段落 (D3: OEM 原件不进库; 词表带出处文件名与出现次数, 不带正文)。
  5. 源 (只读, 不复制): 西门子如东 4.0MW 中文技术资料 / Vestas China Specific 中文集 / GE 培训与流程 / Gamesa 资料。
  6. 用法: python scripts/build_oem_lexicon.py [--roots <json>] [--max-files N] [--out configs/terms/oem_lexicon.yaml]
  7. 产物: 词条 = {term, n (总频), oems[], files (最多 3 个出处)}; 只留在 ≥1 家 OEM 中文文档里出现 ≥MIN_N 次、且长度 2–8 的中文术语。"""
  8. from __future__ import annotations
  9. import argparse, collections, json, os, re, sys, time
  10. from pathlib import Path
  11. ROOT = Path(__file__).resolve().parents[1]
  12. DEFAULT_ROOTS = {
  13. "西门子": ["/Volumes/T5 EVO/风电数据/01_场站SCADA/如东/如东风场数据/西门子4.0技术资料"],
  14. "Vestas": ["/Volumes/WINDDATA/Data sheet/201103Class2/China Specific_Class-2zh", "/Volumes/WINDDATA/Data sheet/201103Class2/China Specific_Class-2"],
  15. "GE": ["/Volumes/T5 EVO/wd/GE tech material", "/Volumes/T5 EVO/ge process"],
  16. "Gamesa": ["/Volumes/WINDDATA/Data sheet/Informations from Gamesa"],
  17. }
  18. CJK = re.compile(r"[一-鿿]{2,8}")
  19. MIN_N = 3
  20. # 领域相关性: 词里含这些字之一才留 (滤掉"公司/项目/附件"类通用词)
  21. DOMAIN_HINT = set("风机组叶片桨距轮毂主轴承齿箱发电变流偏航塔筒基础液压制动刹车润滑油脂冷却散热温升振动轴瓦密封螺栓法兰联轴器"
  22. "编码器传感变频电网并网无功有功功率转速扭矩载荷疲劳裂纹磨损点蚀腐蚀异响泄漏堵塞故障报警停机复位维护检修更换紧固校准标定试验巡检状态监测诊断")
  23. SKIP_DIR = re.compile(r"(^|/)(__MACOSX|\.git)(/|$)")
  24. def texts(p: Path, cap=400_000):
  25. """返回文件的中文文本 (截断). 支持 pdf/docx/xlsx/pptx/txt/html; 失败回空串 (不静默当成功: 调用方计失败数)."""
  26. x = p.suffix.lower()
  27. if x == ".pdf":
  28. import pypdf
  29. r = pypdf.PdfReader(str(p)); out = []
  30. for pg in r.pages[:80]:
  31. out.append(pg.extract_text() or "")
  32. if sum(map(len, out)) > cap: break
  33. return "\n".join(out)
  34. if x in (".docx", ".pptx", ".xlsx"):
  35. import zipfile
  36. with zipfile.ZipFile(p) as z:
  37. names = [n for n in z.namelist() if n.endswith(".xml") and ("document" in n or "slide" in n or "sharedStrings" in n or "sheet" in n)]
  38. buf = []
  39. for n in names[:60]:
  40. buf.append(re.sub(r"<[^>]+>", " ", z.read(n).decode("utf-8", "ignore")))
  41. if sum(map(len, buf)) > cap: break
  42. return " ".join(buf)
  43. if x in (".txt", ".htm", ".html", ".csv"):
  44. return re.sub(r"<[^>]+>", " ", p.read_text(encoding="utf-8", errors="ignore")[:cap])
  45. return ""
  46. def main():
  47. ap = argparse.ArgumentParser(); ap.add_argument("--roots"); ap.add_argument("--max-files", type=int, default=400); ap.add_argument("--out", default="configs/terms/oem_lexicon.yaml"); a = ap.parse_args()
  48. roots = json.loads(Path(a.roots).read_text(encoding="utf-8")) if a.roots else DEFAULT_ROOTS
  49. per_oem = collections.defaultdict(collections.Counter); src = collections.defaultdict(set); stat = {}
  50. for oem, dirs in roots.items():
  51. n_files = n_ok = n_fail = 0; t0 = time.time()
  52. for d in dirs:
  53. base = Path(d)
  54. if not base.exists(): continue
  55. for p in sorted(base.rglob("*")):
  56. if n_files >= a.max_files: break
  57. if not p.is_file() or p.name.startswith("._") or SKIP_DIR.search(str(p)): continue
  58. if p.suffix.lower() not in (".pdf", ".docx", ".pptx", ".xlsx", ".txt", ".htm", ".html", ".csv"): continue
  59. n_files += 1
  60. try: t = texts(p)
  61. except Exception: n_fail += 1; continue
  62. if not t: continue
  63. hits = CJK.findall(t)
  64. if not hits: continue
  65. n_ok += 1
  66. for w in hits:
  67. if any(c in DOMAIN_HINT for c in w):
  68. per_oem[oem][w] += 1
  69. if len(src[w]) < 3: src[w].add(f"{oem}/{p.name}")
  70. stat[oem] = dict(files_scanned=n_files, files_with_cn=n_ok, failed=n_fail, secs=round(time.time() - t0, 1), n_terms=len(per_oem[oem]))
  71. print(f" {oem}: 扫 {n_files} 文件, 含中文 {n_ok}, 失败 {n_fail}, 术语 {len(per_oem[oem])}, {stat[oem]['secs']} s", flush=True)
  72. total = collections.Counter()
  73. for oem, c in per_oem.items(): total.update(c)
  74. items = []
  75. for w, n in total.most_common():
  76. oems = sorted(o for o in per_oem if per_oem[o][w])
  77. if n < MIN_N: continue
  78. items.append(dict(term=w, n=n, oems=oems, per_oem={o: per_oem[o][w] for o in oems}, sources=sorted(src[w])[:3]))
  79. out = ROOT / a.out; out.parent.mkdir(parents=True, exist_ok=True)
  80. import yaml
  81. out.write_text(yaml.safe_dump(dict(schema="oem_lexicon/v0", built=time.strftime("%Y-%m-%d"), note="只存术语与频次+出处文件名, 不存原文 (D3)", min_n=MIN_N, stat=stat, n=len(items), items=items), allow_unicode=True, sort_keys=False), encoding="utf-8")
  82. print(f"词库 {len(items)} 条 → {out}")
  83. if __name__ == "__main__":
  84. main()