pages_sweep.py 15 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. r"""观澜各页面「有没有输出」体检(用户令 2026-09-19)。
  4. 背景: 清过产物 + 从 `data/raw` 重算之后, 页面出现"没有输出"的现象。本器**逐页面(含嵌套页)**走一遍
  5. 网关入口(默认 http://127.0.0.1:28084), 对每一页给出:
  6. HTTP 码 · 字节数 · 判定(有输出 / 空 / 缺件 / 降级 / 报错 / 静态资源)· 证据(页面里那句话)· 疑似原因
  7. 覆盖面:
  8. · 门户 `/`(含内嵌面板是否有内容)· `/healthz` `/api/version` `/ops`
  9. · `/release/*` 本体发布层(随包件, 清产物后必缺)
  10. · `/detail/*` 综合详细分析工作台(v2 的 10 个页签取数接口 + 经典嵌套页 `/turbine/<台>` `/problem/<台>/<系统>`)
  11. · `/cms/*` CMS 振动诊断(总览/逐台页/报告页)
  12. · `/sim/` `/sim/sys/` `/viewer/` 三个静态模块的入口
  13. · `release/` 下**随包静态页**(取数单等, 按入口页里出现的链接发现)
  14. 判定口径(不猜): 先看 HTTP 与字节数; 再看**页面自己认账的那句话**(缺件/无产物/加载失败/…);
  15. JSON 则解析顶层 `err/error/no_products/no_report` 与空数组。每条判定都带证据片段, 便于人工复核。
  16. 用法:
  17. python scripts/pages_sweep.py # 控制台摘要 + 写 UTF-8 报告
  18. python scripts/pages_sweep.py --base http://127.0.0.1:28084
  19. python scripts/pages_sweep.py --write docs/页面输出体检_v0.1.md
  20. """
  21. from __future__ import annotations
  22. try:
  23. from app_common.app_common_guanlan.api import install_root as _install_root
  24. except ImportError: # 理论不可达;包结构异常时回退到按位置上跳
  25. from pathlib import Path as _P
  26. def _install_root(_f): return _P(_f).resolve().parents[4]
  27. import argparse
  28. import json
  29. import pathlib
  30. import re
  31. import sys
  32. import urllib.error
  33. import urllib.parse
  34. import urllib.request
  35. from concurrent.futures import ThreadPoolExecutor
  36. ROOT = _install_root(__file__)
  37. sys.path.insert(0, str(ROOT))
  38. sys.path.insert(0, str(ROOT / 'scripts'))
  39. from src import paths as P # noqa: E402
  40. # ── 页面清单 ────────────────────────────────────────────────────────────────────
  41. # kind: page=人看的页面 · api=取数接口 · static=静态资源
  42. # why : 这一页的产物依赖(用于把"空"归因到"缺哪件")
  43. PAGES: list[dict] = [
  44. dict(path='/', kind='page', why='门户单文件 (release/portal.html), 内嵌结论段由产物渲染',
  45. dep='guanlan/derived/portal_claims.json'),
  46. dict(path='/healthz', kind='api', why='模块探活'),
  47. dict(path='/api/version', kind='api', why='版本/契约指纹'),
  48. dict(path='/ops', kind='page', why='运维控制台 (网关自提供)'),
  49. dict(path='/release/', kind='api', why='本体发布层 manifest (随包件 ontology/release_*)'),
  50. dict(path='/release/r1/manifest.json', kind='api', why='本体发布层 r1', dep='ontology/release_r1/manifest.json'),
  51. dict(path='/release/r2/manifest.json', kind='api', why='本体发布层 r2', dep='ontology/release_r2/manifest.json'),
  52. # /detail 工作台: v2 页签底数 + 各自取数
  53. dict(path='/detail/', kind='page', why='v2 工作台外壳 (10 页签)'),
  54. dict(path='/api/fleet?win=2026%E5%B9%B4', kind='api', why='全部页签底数: windscada/*.parquet + ontology', dep='windscada/temp_monthly.parquet'),
  55. dict(path='/api/curves', kind='api', why='特性曲线页', dep='windscada/curve_lenses.parquet'),
  56. dict(path='/api/vibcms', kind='api', why='振动评估页: windcms/报告_CMS*.md', dep='windcms/报告_CMS振动状态评估报告_2026-09-19.md'),
  57. dict(path='/api/maint_survey', kind='api', why='系统自查页: ontology/objects.json', dep='ontology/objects.json'),
  58. dict(path='/api/ont_chain?kind=tree&system=%E9%BD%BF%E8%BD%AE%E7%AE%B1', kind='api', why='决策链页', dep='ontology/objects.json'),
  59. dict(path='/api/facts', kind='api', why='问答/报告页: guanlan/ 事实契约 (人裁底稿派生)', dep='guanlan/derived/detail_cards.json'),
  60. dict(path='/api/ask_status', kind='api', why='问答页: 本机模型'),
  61. # 经典嵌套页(/detail/<rest> → 组件根路径)
  62. dict(path='/detail/turbine/WTG01', kind='page', why='逐台页 (经典)'),
  63. dict(path='/detail/problem/WTG01/%E5%8F%98%E6%A1%A8%E7%B3%BB%E7%BB%9F', kind='page', why='问题页 (经典)'),
  64. dict(path='/api/turbine?t=WTG01&win=2026%E5%B9%B4', kind='api', why='逐台下钻浮层'),
  65. dict(path='/api/problem?t=WTG01&sys=%E5%8F%98%E6%A1%A8%E7%B3%BB%E7%BB%9F&win=2026%E5%B9%B4',
  66. kind='api', why='问题下钻浮层'),
  67. # /cms CMS 振动诊断
  68. dict(path='/cms/', kind='page', why='CMS 首页 (windcms/index.html)', dep='windcms/index.html'),
  69. dict(path='/cms/overview.html', kind='page', why='CMS 总览', dep='windcms/overview.html'),
  70. dict(path='/cms/turbines/WTG01.html', kind='page', why='CMS 逐台页', dep='windcms/turbines/WTG01.html'),
  71. dict(path='/cms/report.md', kind='page', why='CMS 结构化层报告', dep='windcms/report.md'),
  72. # 静态模块入口
  73. dict(path='/sim/', kind='page', why='仿真与回放 (随包资产)'),
  74. dict(path='/sim/sys/', kind='page', why='仿真·四系统合页 (随包资产)'),
  75. dict(path='/viewer/', kind='page', why='三维拆装工作台 (随包资产)'),
  76. ]
  77. # ★2026-09-19 收紧: 只认"这一页确实没数据"的**具体话术**。首版把 不可用/缺件/无数据 这类通用词也算命中,
  78. # 结果门户图例、i18n 词表、运维页说明文字都被算成"缺件" ⇒ 噪声淹掉真问题。
  79. EMPTY_MARKERS = [
  80. ('无产物', '页面自报"无产物"'), ('产物未生成', '页面自报"产物未生成"'),
  81. ('还没有生成', '页面自报"还没有生成"'), ('本体库缺失', '本体库缺失'),
  82. ('报告解析失败', '报告解析失败'), ('模块不可用: ', '网关的下线页'),
  83. ('加载失败', '页面自报加载失败'), ('读取失败', '页面自报读取失败'),
  84. ('Traceback', '页面里带 traceback'), ('no_products', '接口回 no_products'),
  85. ('not found', '404/未找到'),
  86. ]
  87. SPA_SHELLS = {'/', '/detail/'} # 单页应用壳: 正文由 JS 取数渲染, 不能按"可见文本"判空
  88. def fetch(base: str, path: str, timeout: int = 60):
  89. url = base + path
  90. try:
  91. with urllib.request.urlopen(url, timeout=timeout) as r:
  92. return r.status, r.headers.get('Content-Type', ''), r.read()
  93. except urllib.error.HTTPError as e:
  94. return e.code, e.headers.get('Content-Type', '') if e.headers else '', e.read()
  95. except Exception as e:
  96. return None, f'{type(e).__name__}', str(e).encode('utf-8')
  97. _TAG = re.compile(r'<[^>]+>')
  98. _SCRIPT = re.compile(r'<(script|style)\b.*?</\1>', re.S | re.I)
  99. _COMMENT = re.compile(r'<!--.*?-->', re.S)
  100. def visible_text(html: str) -> str:
  101. """页面**可见文本**(去掉 script/style/注释/标签)。
  102. ★为什么必须先去脚本: 首版直接在整页文本里找"不可用/undefined/缺件"这类词, 命中的全是 JS 里的
  103. i18n 词表与 CSS 类名(例如逐台页里 `'监测不可用':'var(--unknown)'`)⇒ **好页面被误判成缺件**。
  104. 判定只许看人能看到的那部分。
  105. """
  106. t = _SCRIPT.sub(' ', html)
  107. t = _COMMENT.sub(' ', t)
  108. t = _TAG.sub(' ', t)
  109. return re.sub(r'\s+', ' ', t).strip()
  110. def judge(path: str, kind: str, code, ctype: str, body: bytes) -> dict:
  111. raw = body.decode('utf-8', 'replace')
  112. n = len(body)
  113. is_html = ('html' in (ctype or '').lower()) or raw.lstrip()[:1] == '<'
  114. txt = visible_text(raw) if is_html else raw
  115. hits = [why for mk, why in EMPTY_MARKERS if mk in txt]
  116. err_msg = ''
  117. if 'json' in (ctype or '') and n < 400000:
  118. try:
  119. o = json.loads(raw)
  120. if isinstance(o, dict):
  121. err_msg = str(o.get('err') or o.get('error') or '')
  122. if o.get('no_products') or o.get('no_report'):
  123. hits.append('接口自报无产物')
  124. empties = [k for k, v in o.items() if isinstance(v, list) and not v]
  125. if empties and len(empties) >= max(1, len(o) // 2):
  126. hits.append('顶层多数数组为空: ' + ','.join(empties[:4]))
  127. except Exception:
  128. pass
  129. if code is None:
  130. verdict = '连不上'
  131. elif code >= 500:
  132. verdict = '报错'
  133. elif code >= 400:
  134. verdict = '报错' if code != 404 else '未找到'
  135. elif kind == 'static':
  136. verdict = '静态资源'
  137. elif n < 200:
  138. verdict = '空'
  139. elif hits:
  140. verdict = '缺件/降级'
  141. elif path in SPA_SHELLS:
  142. verdict = '有输出(单页壳)' # 壳本身只有标题, 正文由 JS 从 /api/* 取 —— 判空要看它的接口
  143. elif is_html and len(txt) < 120 and n < 60000:
  144. # ★阈值要带字节数: 200 KB 的逐台/问题页可见文本本来就少(正文全是 JS 渲染的),
  145. # 只看"可见文本 <120 字"会把它们误判成空页(2026-09-19 实逮)。
  146. verdict = '空(可见文本极少)'
  147. else:
  148. verdict = '有输出'
  149. ev = err_msg[:90] or (hits[0] if hits else txt[:70])
  150. return dict(verdict=verdict, code=code, n=n, ctype=(ctype or '')[:40], hits=hits[:3],
  151. evidence=ev, visible=len(txt))
  152. CAUSE_RUNNING = '重算进行中(跑完自愈)'
  153. CAUSE_NOGEN = '该族无生成端(重算也补不出)'
  154. CAUSE_HUMAN = '依赖人裁底稿(非机械可算)'
  155. CAUSE_EXTERNAL = '外部组件(本机模型)'
  156. CAUSE_STATIC = '随包静态资产(与重算无关)'
  157. CAUSE_OK = '有输出'
  158. def _families() -> tuple:
  159. """→ (族表, 是否正在重算)。族表用来判"缺的这一族有没有生成端"; 在跑与否决定能不能自愈。"""
  160. fams = []
  161. try:
  162. from .products_reverse_audit import FAMILIES
  163. fams = FAMILIES
  164. except Exception:
  165. pass
  166. running = False
  167. try:
  168. from src import opsjob
  169. running = bool(opsjob.current().get('running'))
  170. except Exception:
  171. pass
  172. return fams, running
  173. def family_of(fams, rel: str):
  174. import fnmatch
  175. for fam in fams:
  176. pats = [fam.get('glob')] if isinstance(fam.get('glob'), str) else list(fam.get('glob') or [])
  177. for g in pats:
  178. if '{' in g:
  179. head, tail = g.split('{', 1)
  180. opts, rest = tail.split('}', 1)
  181. exp = [head + o + rest for o in opts.split(',')]
  182. else:
  183. exp = [g]
  184. for pt in exp:
  185. if '**' not in pt and rel.count('/') != pt.count('/'):
  186. continue
  187. if rel == pt or fnmatch.fnmatch(rel, pt):
  188. return fam
  189. return None
  190. def cause_of(path: str, verdict: str, dep: str | None, fams, running: bool) -> str:
  191. """把"空/报错"翻成原因: 依赖族有没有生成端 → 会不会自愈。判据只有事实, 不猜。"""
  192. if verdict == '有输出':
  193. return CAUSE_OK
  194. if verdict == '静态资源':
  195. return CAUSE_STATIC
  196. if path.startswith('/local-ai') or '/ask' in path:
  197. return CAUSE_EXTERNAL
  198. f = family_of(fams, dep) if dep else None
  199. if f is None:
  200. return CAUSE_RUNNING if running else '依赖件不在位(放原始件后重算)'
  201. if f.get('kind') in ('shipped', 'human') and not f.get('gen'):
  202. # 本体发布层/带版本交付件 = 研发出件(不是人裁底稿,别把两类混成一句)
  203. if f.get('id') in ('ontology_releases', 'windscada_pages', 'tcm_replay'):
  204. return '包内无生成端(研发出件)'
  205. return CAUSE_HUMAN if f.get('id') in ('guanlan_contract', 'sop_workspace', 'paradigm_r1',
  206. 'human_deliverables') else CAUSE_NOGEN
  207. return CAUSE_RUNNING if running else '依赖件不在位(重算未覆盖到)'
  208. def discover_links(base: str, prefix: str, html: bytes, limit: int = 25) -> list[str]:
  209. """从页面里发现**同前缀**的嵌套页(href/src),用于"含嵌套页面"的覆盖。"""
  210. txt = html.decode('utf-8', 'replace')
  211. got, seen = [], set()
  212. for m in re.finditer(r'(?:href|src)=["\']([^"\'#?]+)["\']', txt):
  213. u = m.group(1)
  214. if not u.startswith('/') or not u.startswith(prefix):
  215. continue
  216. if u.endswith(('.js', '.css', '.png', '.jpg', '.svg', '.woff', '.woff2', '.ico', '.map')):
  217. continue
  218. if u in seen or '${' in u: # JS 模板字面量不是页面
  219. continue
  220. seen.add(u)
  221. got.append(u)
  222. if len(got) >= limit:
  223. break
  224. return got
  225. def main() -> int:
  226. ap = argparse.ArgumentParser()
  227. ap.add_argument('--base', default='http://127.0.0.1:28084')
  228. ap.add_argument('--write', default=None)
  229. ap.add_argument('--links', type=int, default=12, help='每个模块再抽几个嵌套页')
  230. a = ap.parse_args()
  231. tasks = list(PAGES)
  232. # 嵌套页: 先抓每个模块入口, 从中发现同前缀链接
  233. for prefix in ('/detail', '/cms', '/sim', '/sim/sys', '/viewer', '/ops'):
  234. code, ct, body = fetch(a.base, prefix + '/')
  235. if not body:
  236. continue
  237. for u in discover_links(a.base, prefix, body, a.links):
  238. tasks.append(dict(path=u, kind='page', why=f'{prefix} 内的嵌套页 (从入口页发现)'))
  239. fams, running = _families()
  240. with ThreadPoolExecutor(max_workers=8) as ex:
  241. res = list(ex.map(lambda t: (t, fetch(a.base, t['path'])), tasks))
  242. rows = []
  243. for t, (code, ctype, body) in res:
  244. j = judge(t['path'], t['kind'], code, ctype, body)
  245. j.update(path=t['path'], kind=t['kind'], why=t['why'], dep=t.get('dep'))
  246. j['cause'] = cause_of(t['path'], j['verdict'], t.get('dep'), fams, running)
  247. rows.append(j)
  248. order = {'报错': 0, '未找到': 1, '空': 2, '缺件/降级': 3, '连不上': 4, '有输出': 5, '静态资源': 6}
  249. rows.sort(key=lambda r: (order.get(r['verdict'], 9), r['path']))
  250. L = [f'# 观澜页面输出体检 · {a.base}', '',
  251. f'共 {len(rows)} 页(含从各模块入口发现的嵌套页)· 判定口径: HTTP + 字节数 + 页面自己认账的那句话',
  252. f'· 重算是否正在跑: **{"是(产物会随步骤长回来)" if running else "否"}**', '',
  253. '| 判定 | 路径 | 码 | 字节 | 原因 | 证据 | 产物依赖 |', '|---|---|---:|---:|---|---|---|']
  254. for r in rows:
  255. L.append(f"| {r['verdict']} | `{r['path']}` | {r['code']} | {r['n']:,} | {r['cause']} | "
  256. f"{str(r['evidence'])[:60]} | {r['why']} |")
  257. rep = '\n'.join(L)
  258. if a.write:
  259. pathlib.Path(a.write).write_text(rep + '\n', encoding='utf-8')
  260. from collections import Counter
  261. cnt = Counter(r['verdict'] for r in rows)
  262. print('判定分布: ' + ' · '.join(f'{k}={v}' for k, v in cnt.most_common()))
  263. for r in rows:
  264. if r['verdict'] not in ('有输出', '静态资源'):
  265. print(f" [{r['verdict']}] {r['path']} HTTP {r['code']} {r['n']:,}B {str(r['evidence'])[:70]}")
  266. if a.write:
  267. print(f'已写 {a.write}')
  268. return 0
  269. if __name__ == '__main__':
  270. sys.exit(main())