| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- """观澜云端版 · 方案 A 中文单文件 (用户裁 2026-09-07 "中文版上云走 A: 脱敏面孔"): 部署包页 → 脱敏 → 云端契约段 → 回扫 → 打包目录.
- 流程 (可复跑, 每步有证据):
- 1. 挖空 base64 图片块 (`data:…;base64,…` → 占位符; 字节不含可识别文本, 不参与扫描/替换, 完后回填)
- 2. src/windscada/deid_public.scrub 全局一致替换 (场名/OEM/机型/业主/供应商/文档名/内部系统名)
- 3. 注入云端契约段 (scripts/guanlan_portal_inject_claims.py, 数据源 = 脱敏面孔 cloud/portal_claims.json, 与内部契约同 sha)
- 4. 回扫: deid_public.LEAK_RE + guanlan_cloud_face.EXTRA_RE (绝对路径/邮箱/手机/身份证/坐标/密钥) 任一命中 → 不出件 (exit 2);
- windscada_cloud_pack.GATES 只报 (中文版口径: 机组号/台数/厂商名待裁); 重复 id / 锚点缺目标 任一 → 不出件
- 5. 打包目录 cloud/_pkg/ (index.html + 五个脱敏面孔 JSON/MD + SHA256SUMS + README) —— 目录 gitignored (20 MB), 只登记 sha 于 cloud/scan_report_page.md
- 铁律: "替换过了"不算数, 只有回扫零命中算; 上传仍须过 scripts/deploy_gate_check.py --farm rudong (PB-1) 与用户显式批准, 本脚本不上传.
- 用法: build [--src <html>] | check
- """
- import argparse, datetime as dt, hashlib, json, re, subprocess, sys, time
- from collections import Counter
- from pathlib import Path
- ROOT = Path(__file__).resolve().parents[1]
- sys.path.insert(0, str(ROOT / "src")); sys.path.insert(0, str(ROOT / "scripts"))
- from windscada import deid_public as DP # noqa: E402
- import guanlan_cloud_face as CF # noqa: E402
- import windscada_cloud_pack as CP # noqa: E402
- import guanlan_portal_inject_claims as INJ # noqa: E402
- CLOUD = ROOT / "outputs/rudong/guanlan/cloud"; BUILD = CLOUD / "_build"; PKG = CLOUD / "_pkg"; REPORT = CLOUD / "scan_report_page.md"
- DEF_SRC = Path.home() / "Desktop/观澜中文系统_详细分析_第一版部署包_20260906 2/观澜中文系统_详细分析.html"
- FACE_FILES = ("portal_claims.json", "detail_cards.json", "qa_refs.json", "report_summary.md", "claims_public.json", "cloud_manifest.json")
- _B64 = re.compile(r"data:[a-zA-Z0-9.+/-]+;base64,[A-Za-z0-9+/=]+")
- sha = lambda b: hashlib.sha256(b).hexdigest()
- _LOCAL = re.compile(r"https?://(?:127\.0\.0\.1|localhost):(\d+)/([^\"'\s<>]*)")
- _NOTE = '<p class="offline-note" style="margin:8px 0;padding:8px 12px;border:1px dashed #9DACB7;border-radius:6px;color:#5B6C78;font-size:13px">本模块 (本机服务) 仅离线版提供, 云端未部署。</p>'
- def cloud_links(t: str) -> str:
- """本机离线地址 → 云端地址 (2026-09-07 用户报 "云上三维拆装打不开": 中文页"三维拆装"按钮把 iframe 指向 127.0.0.1:54292, 云端点了必死).
- 覆盖 href / iframe src / 内联 JS 单引号 / srcdoc 转义 (") 四种写法:
- 54292 unit-workbench.html?component=X → /turbine-explorer?component=X (云端三维页, 实测支持 component 参数)
- 其余本机端口 (8020 CMS / 8791·8792 仿真 / 8033 / 18084 治理清单) 云端未部署 → about:blank; <a> 去 href 变不可点 + 提示; <iframe> 前加可见说明."""
- def rep(m):
- port, rest = m.group(1), m.group(2)
- if port in ("54292", "54312"): # 本机三维工作台 (54312 = Codex 当前项目预览) → 云端中文版 (2026-09-07 用户令 "接入我刚完成的中文版"); 英文版仍在 /turbine-explorer
- comp = re.search(r"component=([A-Za-z0-9._-]+)", rest); return "/turbine-explorer-zh" + (("?component=" + comp.group(1)) if comp else "")
- if port in ("8792", "8791"):
- import guanlan_cloud_sims as GS; key = f"http://127.0.0.1:{port}/" + rest.split("#")[0].split("?")[0]
- slug = GS.SIMS.get(key) or GS.SIMS.get(key.replace("sc1_sim.html", "sc1_sim_demo.html"))
- if slug: return f"/sim/{slug}.html" + (("#" + rest.split("#", 1)[1]) if "#" in rest else "")
- if port == "18084" and rest.startswith("观澜"): return "/zh" + (("#" + rest.split("#", 1)[1]) if "#" in rest else "")
- return "about:blank"
- t = _LOCAL.sub(rep, t)
- t = re.sub(r'href="about:blank"', 'data-offline-only="1" title="本模块仅离线版提供 (云端未部署)"', t)
- t = re.sub(r"(<iframe\b[^>]*\bsrc=\"about:blank\"[^>]*>)", lambda m: _NOTE + m.group(1), t)
- # 页内 iframe 覆盖层 (onclick 设 f.src) : 云端三维页带 X-Frame-Options DENY / frame-ancestors none, 同源 iframe 也被拒 → "内嵌打开"改成新标签直开; 指向 about:blank 的覆盖层按钮改不可点
- t = re.sub(r'href="#" onclick="[^"]*?f\.src=\'(/turbine-explorer(?:-zh)?[^\']*)\'[^"]*"', r'href="\1" target="_blank" rel="noopener"', t)
- t = re.sub(r'(<a class="chip ev" href="/turbine-explorer(?:-zh)?[^"]*" target="_blank" rel="noopener">)内嵌打开工作台 →', r'\1打开工作台 ↗', t)
- t = re.sub(r'href="#" onclick="[^"]*?f\.src=\'about:blank\'[^"]*"', 'data-offline-only="1" title="本模块仅离线版提供 (云端未部署)"', t)
- return t
- def script_syntax_errors(t: str, limit=40):
- """每段内联脚本 (主文档 + 所有 srcdoc 内嵌文档) 过 node 语法检查 (new Function). 2026-09-07 实逮: 脱敏把 `document.documentElement` 改成
- `[technical document]umentElement`, 五个仿真页图区全空, 而回扫闸只认残留不认脚本坏 → 这里兜底. node 不在时只报不拦."""
- import html as _h, shutil, subprocess as _sp
- node = shutil.which("node") or "/opt/homebrew/bin/node"
- if not Path(node).exists(): return ["(node 不在, 跳过脚本语法闸)"], False
- docs = [t] + [_h.unescape(m.group(1)) for m in re.finditer(r'srcdoc="([^"]*)"', t)]
- errs = []
- for di, d in enumerate(docs):
- for si, m in enumerate(re.finditer(r"<script(?![^>]*\bsrc=)[^>]*>(.*?)</script>", d, re.S)):
- body = m.group(1)
- if not body.strip(): continue
- r = _sp.run([node, "-e", "let s='';process.stdin.on('data',c=>s+=c).on('end',()=>{try{new Function(s)}catch(e){console.log(String(e).split('\\n')[0]);process.exit(3)}})"], input=body, capture_output=True, text=True)
- if r.returncode != 0:
- bad = re.search(r"\[technical document\]|\[OEM [a-z]+\]|\[reference table\]", body); errs.append(f"doc{di}/script{si}: {r.stdout.strip()[:80]}" + (f" (含脱敏占位 @{bad.start()})" if bad else ""))
- if len(errs) >= limit: return errs, True
- return errs, True
- CHROME = "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome"
- def render_check(t: str, extra_pages, budget_ms=8000):
- """渲染闸 (用户令 2026-09-07 "建立更新自适应机制"): 无头 Chrome 真加载主页 + 每个 srcdoc 内嵌页 + 每个仿真页, 收控制台 ERROR (含 Uncaught) + 截图空白判定.
- 只看"页面自己的错" (file:// 下无 CSP, 外链字体能加载, 不会误报 CSP 拒绝). Chrome 不在 → 报一条不拦."""
- import html as _h, subprocess as _sp, tempfile
- if not Path(CHROME).exists(): return ["(Chrome 不在, 跳过渲染闸)"]
- bad = []; tmp = Path(tempfile.mkdtemp(prefix="guanlan_render_"))
- pages = [("main", t)] + [(f"srcdoc{i}", _h.unescape(m.group(1))) for i, m in enumerate(re.finditer(r'srcdoc="([^"]*)"', t))]
- files = []
- for name, body in pages: f = tmp / f"{name}.html"; f.write_text(body, encoding="utf-8"); files.append((name, f))
- files += [(f"sim/{p.stem}", p) for p in extra_pages if p.is_file()]
- try:
- from PIL import Image, ImageStat
- except Exception: Image = None
- for name, f in files:
- png = tmp / (name.replace("/", "_") + ".png")
- r = _sp.run([CHROME, "--headless=new", "--no-first-run", "--use-angle=swiftshader", "--enable-unsafe-swiftshader", "--ignore-gpu-blocklist", "--enable-logging=stderr", "--v=0", f"--virtual-time-budget={budget_ms}", "--window-size=1400,1000", f"--screenshot={png}", f.as_uri()], capture_output=True, text=True, timeout=180)
- errs = [l for l in r.stderr.splitlines() if ":ERROR:CONSOLE" in l or "Uncaught" in l]
- errs = [e for e in errs if "net::ERR_" not in e and "fonts.g" not in e] # 离线/无网时字体等资源加载失败不算页面错
- if errs: bad.append(f"{name}: {len(errs)} 条脚本错误, 首条 {errs[0].split('CONSOLE')[-1][:110]}")
- if Image is not None and png.is_file():
- im = Image.open(png).convert("L"); st = ImageStat.Stat(im)
- if st.stddev[0] < 3: bad.append(f"{name}: 截图近乎空白 (灰度 std {st.stddev[0]:.1f})")
- elif not png.is_file(): bad.append(f"{name}: Chrome 未出截图 (rc={r.returncode})")
- return bad
- def build(src: Path):
- t0 = time.time(); raw = src.read_text(encoding="utf-8"); blobs = []
- def hole(m): blobs.append(m.group(0)); return f"@@BLOB{len(blobs)-1}@@"
- txt = _B64.sub(hole, raw); before, before_cp = CF.scan(txt), CP.audit(txt)
- t = cloud_links(DP.scrub(txt))
- BUILD.mkdir(parents=True, exist_ok=True); tmp = BUILD / "_tmp.html"; tmp.write_text(t, encoding="utf-8")
- INJ.PC = CLOUD / "portal_claims.json"; INJ.DC = CLOUD / "detail_cards.json"; INJ.CONTRACT = CLOUD / "claims_public.json" # 云端只认脱敏面孔三件, 不读内部契约
- INJ.inject(tmp, tmp); t = tmp.read_text(encoding="utf-8"); tmp.unlink()
- after, after_cp, info = CF.scan(t), CP.audit(t), CF.info(t)
- ids = Counter(re.findall(r'\sid="([^"]+)"', t)); dup = sorted(i for i, c in ids.items() if c > 1)
- hrefs = set(re.findall(r'href="#([^"]+)"', t)); missing = sorted(h for h in hrefs if h not in ids)
- full = re.sub(r"@@BLOB(\d+)@@", lambda m: blobs[int(m.group(1))], t); out = BUILD / "观澜中文系统_详细分析_cloud_candidate.html"; out.write_text(full, encoding="utf-8")
- local_left = len(re.findall(r"https?://(?:127\.0\.0\.1|localhost)[:/]", t))
- import guanlan_cloud_sims as GS; sim_done, sim_problems = GS.build()
- js_errs, js_checked = script_syntax_errors(t)
- for jsf in (PKG / "sim").glob("*.js"): # 独立 .js (sc1 参数文件) 也过 node 语法闸
- e2, _ = script_syntax_errors(f"<script>{jsf.read_text(encoding='utf-8')}</script>"); js_errs += [f"sim/{jsf.name}: {x}" for x in e2]
- render_bad = render_check(t, [PKG / "sim" / f"{s}.html" for s, _ in sim_done if not s.endswith(".js")])
- problems = ([f"回扫残留 {k}×{v[0]} {v[1]}" for k, v in after.items()] + ([f"本机地址残留 {local_left} 处"] if local_left else []) + ([f"内联脚本语法错 {len(js_errs)} 段: {js_errs[:3]}"] if js_checked and js_errs else []) + [f"仿真页 {x}" for x in sim_problems] + [f"渲染闸 {x}" for x in render_bad] + ([f"重复 id {dup}"] if dup else []) + ([f"锚点缺目标 {missing[:8]}"] if missing else []))
- git = subprocess.run(["git", "-C", str(ROOT), "rev-parse", "--short", "HEAD"], capture_output=True, text=True).stdout.strip()
- pkg_sums = {}
- if not problems:
- PKG.mkdir(parents=True, exist_ok=True)
- for old in PKG.glob("*"):
- if old.is_file(): old.unlink()
- (PKG / "index.html").write_bytes(full.encode("utf-8"))
- for n in FACE_FILES:
- if (CLOUD / n).is_file(): (PKG / n).write_bytes((CLOUD / n).read_bytes())
- (PKG / "README.md").write_text(f"# 观澜云端版 · 方案 A 中文包 (脱敏面孔)\n\n构建 {dt.datetime.now():%Y-%m-%d %H:%M} · git {git} · 契约 {json.loads((CLOUD / 'cloud_manifest.json').read_text(encoding='utf-8'))['contract']['contract_sha256'][:16]}\n"
- "index.html = 部署包中文单文件经 deid_public 全局替换 + 云端契约段; 其余 = scripts/guanlan_cloud_face.py 五产物。\n"
- "上传前: python scripts/deploy_gate_check.py --farm rudong --report-html outputs/rudong/guanlan/cloud/_pkg/index.html (PB-1) + 用户显式批准; 部署草案见 deploy/guanlan_cloud/。\n", encoding="utf-8")
- pkg_sums = {str(p.relative_to(PKG)): sha(p.read_bytes()) for p in sorted(PKG.rglob("*")) if p.is_file() and p.name != "SHA256SUMS"}
- (PKG / "SHA256SUMS").write_text("".join(f"{v} {k}\n" for k, v in pkg_sums.items()), encoding="utf-8")
- rep = ["# 云端中文单文件 (方案 A, 用户裁 2026-09-07) · 扫描报告 (scripts/guanlan_cloud_page.py 自动生成)", "",
- f"源 `{src.name}` sha {sha(raw.encode())[:16]} ({len(raw)//1024} KB) → 候选 sha {sha(full.encode())[:16]} ({len(full)//1024} KB); 耗时 {time.time()-t0:.0f}s; base64 块 {len(blobs)} 个原样回填; git {git}。**候选/打包目录 gitignored, 只登记 sha。**", "",
- "| 闸 | 脱敏前 | 脱敏后 |", "|---|---|---|",
- f"| deid_public.LEAK_RE + §C (绝对路径/邮箱/手机/身份证/坐标/密钥) — block | {', '.join(f'{k}×{v[0]}' for k, v in before.items()) or 0} | {', '.join(f'{k}×{v[0]}' for k, v in after.items()) or 0} |",
- f"| windscada_cloud_pack.GATES — 只报 (中文版口径待裁) | {before_cp or 0} | {after_cp or 0} |", f"| 机组号/台数 — 只报 | — | {info or '—'} |",
- f"| 内部锚点 — block | 锚点 {len(hrefs)} | 缺目标 {missing or 0} · 重复 id {dup or 0} |", "",
- f"| 本机地址残留 — block | — | {local_left} |", f"| 内联脚本 node 语法闸 — block | — | {('错 ' + str(len(js_errs))) if js_errs else ('PASS' if js_checked else '跳过')} |",
- f"| 仿真页打包 (本机 8792/8791 → /sim/) — block | — | {len(sim_done)} 件, 问题 {sim_problems or 0} |", f"| 无头 Chrome 渲染闸 (主页+内嵌+仿真, 脚本错/空白) — block | — | {render_bad or 'PASS'} |", "",
- "判定: " + ("**PASS** — 已出打包目录 `cloud/_pkg/` (index.html + 面孔五件 + SHA256SUMS + README)" if not problems else "**FAIL** (不出件): " + "; ".join(problems)), ""]
- if pkg_sums: rep += ["## 打包目录 SHA-256", ""] + [f"- {k}: {v}" for k, v in pkg_sums.items()] + [""]
- rep += ["注: ① 只报不拦项 = SKF/FAG 等部件厂商名 / 机组号 / 台数 / 判据门槛, 按 2026-09-01 中文版口径 (给业主本人看) 不阻塞, 是否代号化待裁; ② 上传另有 PB-1 总闸 (现 BLOCKED 存量债 75 条) 与用户批准, 本脚本不上传。"]
- REPORT.write_text("\n".join(rep) + "\n", encoding="utf-8"); print("\n".join(rep[:12]))
- return 0 if not problems else 2
- def check():
- """打包目录与报告对账: SHA256SUMS 逐文件核 + index.html 回扫零命中 + 面孔 JSON 内嵌契约 sha == 当前契约."""
- bad = []
- if not (PKG / "SHA256SUMS").is_file(): return ["_pkg/SHA256SUMS 不存在 (先 build)"]
- for line in (PKG / "SHA256SUMS").read_text(encoding="utf-8").splitlines():
- h, n = line.split(" ", 1); p = PKG / n
- if not p.is_file(): bad.append(f"{n} 缺失")
- elif sha(p.read_bytes()) != h: bad.append(f"{n} sha ≠ SHA256SUMS (被手改)")
- html = _B64.sub("@@B@@", (PKG / "index.html").read_text(encoding="utf-8"))
- for k, (c, ex) in CF.scan(html).items(): bad.append(f"index.html 残留 {k}×{c} {ex}")
- con_sha = json.loads((ROOT / "outputs/rudong/guanlan/facts_contract_v0.json").read_text(encoding="utf-8"))["contract_sha256"]
- for n in ("portal_claims.json", "detail_cards.json", "qa_refs.json", "claims_public.json"):
- if (PKG / n).is_file() and json.loads((PKG / n).read_text(encoding="utf-8")).get("contract_sha256") != con_sha: bad.append(f"{n} 内嵌契约 sha ≠ 当前契约")
- if f'data-contract="{con_sha}"' not in html: bad.append("index.html 契约段 sha ≠ 当前契约")
- return bad
- if __name__ == "__main__":
- ap = argparse.ArgumentParser(); ap.add_argument("mode", nargs="?", default="check", choices=["build", "check"]); ap.add_argument("--src", default=str(DEF_SRC)); a = ap.parse_args()
- if a.mode == "build": sys.exit(build(Path(a.src)))
- b = check(); print("PASS" if not b else "FAIL", *b[:10], sep="\n "); sys.exit(0 if not b else 2)
|