| 12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- """OEM 中文术语库构建 (用户令 2026-09-08: "找 GE、西门子、维斯塔斯、歌美飒的中文文档, 建立词库, 审核中文表达").
- 只抽**术语与词频**, 不复制原文段落 (D3: OEM 原件不进库; 词表带出处文件名与出现次数, 不带正文)。
- 源 (只读, 不复制): 西门子如东 4.0MW 中文技术资料 / Vestas China Specific 中文集 / GE 培训与流程 / Gamesa 资料。
- 用法: python scripts/build_oem_lexicon.py [--roots <json>] [--max-files N] [--out configs/terms/oem_lexicon.yaml]
- 产物: 词条 = {term, n (总频), oems[], files (最多 3 个出处)}; 只留在 ≥1 家 OEM 中文文档里出现 ≥MIN_N 次、且长度 2–8 的中文术语。"""
- from __future__ import annotations
- import argparse, collections, json, os, re, sys, time
- from pathlib import Path
- ROOT = Path(__file__).resolve().parents[1]
- DEFAULT_ROOTS = {
- "西门子": ["/Volumes/T5 EVO/风电数据/01_场站SCADA/如东/如东风场数据/西门子4.0技术资料"],
- "Vestas": ["/Volumes/WINDDATA/Data sheet/201103Class2/China Specific_Class-2zh", "/Volumes/WINDDATA/Data sheet/201103Class2/China Specific_Class-2"],
- "GE": ["/Volumes/T5 EVO/wd/GE tech material", "/Volumes/T5 EVO/ge process"],
- "Gamesa": ["/Volumes/WINDDATA/Data sheet/Informations from Gamesa"],
- }
- CJK = re.compile(r"[一-鿿]{2,8}")
- MIN_N = 3
- # 领域相关性: 词里含这些字之一才留 (滤掉"公司/项目/附件"类通用词)
- DOMAIN_HINT = set("风机组叶片桨距轮毂主轴承齿箱发电变流偏航塔筒基础液压制动刹车润滑油脂冷却散热温升振动轴瓦密封螺栓法兰联轴器"
- "编码器传感变频电网并网无功有功功率转速扭矩载荷疲劳裂纹磨损点蚀腐蚀异响泄漏堵塞故障报警停机复位维护检修更换紧固校准标定试验巡检状态监测诊断")
- SKIP_DIR = re.compile(r"(^|/)(__MACOSX|\.git)(/|$)")
- def texts(p: Path, cap=400_000):
- """返回文件的中文文本 (截断). 支持 pdf/docx/xlsx/pptx/txt/html; 失败回空串 (不静默当成功: 调用方计失败数)."""
- x = p.suffix.lower()
- if x == ".pdf":
- import pypdf
- r = pypdf.PdfReader(str(p)); out = []
- for pg in r.pages[:80]:
- out.append(pg.extract_text() or "")
- if sum(map(len, out)) > cap: break
- return "\n".join(out)
- if x in (".docx", ".pptx", ".xlsx"):
- import zipfile
- with zipfile.ZipFile(p) as z:
- names = [n for n in z.namelist() if n.endswith(".xml") and ("document" in n or "slide" in n or "sharedStrings" in n or "sheet" in n)]
- buf = []
- for n in names[:60]:
- buf.append(re.sub(r"<[^>]+>", " ", z.read(n).decode("utf-8", "ignore")))
- if sum(map(len, buf)) > cap: break
- return " ".join(buf)
- if x in (".txt", ".htm", ".html", ".csv"):
- return re.sub(r"<[^>]+>", " ", p.read_text(encoding="utf-8", errors="ignore")[:cap])
- return ""
- def main():
- ap = argparse.ArgumentParser(); ap.add_argument("--roots"); ap.add_argument("--max-files", type=int, default=400); ap.add_argument("--out", default="configs/terms/oem_lexicon.yaml"); a = ap.parse_args()
- roots = json.loads(Path(a.roots).read_text(encoding="utf-8")) if a.roots else DEFAULT_ROOTS
- per_oem = collections.defaultdict(collections.Counter); src = collections.defaultdict(set); stat = {}
- for oem, dirs in roots.items():
- n_files = n_ok = n_fail = 0; t0 = time.time()
- for d in dirs:
- base = Path(d)
- if not base.exists(): continue
- for p in sorted(base.rglob("*")):
- if n_files >= a.max_files: break
- if not p.is_file() or p.name.startswith("._") or SKIP_DIR.search(str(p)): continue
- if p.suffix.lower() not in (".pdf", ".docx", ".pptx", ".xlsx", ".txt", ".htm", ".html", ".csv"): continue
- n_files += 1
- try: t = texts(p)
- except Exception: n_fail += 1; continue
- if not t: continue
- hits = CJK.findall(t)
- if not hits: continue
- n_ok += 1
- for w in hits:
- if any(c in DOMAIN_HINT for c in w):
- per_oem[oem][w] += 1
- if len(src[w]) < 3: src[w].add(f"{oem}/{p.name}")
- stat[oem] = dict(files_scanned=n_files, files_with_cn=n_ok, failed=n_fail, secs=round(time.time() - t0, 1), n_terms=len(per_oem[oem]))
- print(f" {oem}: 扫 {n_files} 文件, 含中文 {n_ok}, 失败 {n_fail}, 术语 {len(per_oem[oem])}, {stat[oem]['secs']} s", flush=True)
- total = collections.Counter()
- for oem, c in per_oem.items(): total.update(c)
- items = []
- for w, n in total.most_common():
- oems = sorted(o for o in per_oem if per_oem[o][w])
- if n < MIN_N: continue
- items.append(dict(term=w, n=n, oems=oems, per_oem={o: per_oem[o][w] for o in oems}, sources=sorted(src[w])[:3]))
- out = ROOT / a.out; out.parent.mkdir(parents=True, exist_ok=True)
- import yaml
- out.write_text(yaml.safe_dump(dict(schema="oem_lexicon/v0", built=time.strftime("%Y-%m-%d"), note="只存术语与频次+出处文件名, 不存原文 (D3)", min_n=MIN_N, stat=stat, n=len(items), items=items), allow_unicode=True, sort_keys=False), encoding="utf-8")
- print(f"词库 {len(items)} 条 → {out}")
- if __name__ == "__main__":
- main()
|